diff --git a/docs/.vitepress/config.mts b/docs/.vitepress/config.mts index 6e69ac30..8c39ba05 100644 --- a/docs/.vitepress/config.mts +++ b/docs/.vitepress/config.mts @@ -229,6 +229,9 @@ export default defineConfig({ { text: 'flame_chase', link: '/flows/flame-chase' }, { text: 'rlar', link: '/flows/rlar' }, { text: 'humanize1', link: '/flows/humanize1' }, + { text: '…_agent_cleanup', link: '/flows/agent-cleanup' }, + { text: 'recursive_lean_prover', link: '/flows/recursive-lean-prover' }, + { text: 'aot', link: '/flows/aot' }, ], }, { diff --git a/docs/.vitepress/theme/components/HmzDaemon.vue b/docs/.vitepress/theme/components/HmzDaemon.vue index de76f1f0..a9ba8b17 100644 --- a/docs/.vitepress/theme/components/HmzDaemon.vue +++ b/docs/.vitepress/theme/components/HmzDaemon.vue @@ -335,16 +335,16 @@ function reset() { {{ journal }} complete lines · append as events happen
- state.json - revision {{ stateRevision }} · replace when the flow writes + resume.jsonl + {{ stateRevision }} state writes · each flushed as the flow makes it
daemon.log append failures that no terminal could show

- The terminal screen is not kept here. The journal records the run's shape, state is - what a resumable flow chose to keep, and the backend owns the conversation. + The terminal screen is not kept here. The epic records the run's shape, the engine's + journal what a resumable flow chose to keep, and the backend owns the conversation.

diff --git a/docs/.vitepress/theme/components/HmzFlows.vue b/docs/.vitepress/theme/components/HmzFlows.vue index 86a5de06..c87eaa47 100644 --- a/docs/.vitepress/theme/components/HmzFlows.vue +++ b/docs/.vitepress/theme/components/HmzFlows.vue @@ -13,9 +13,9 @@ import { FLOWS, type Place } from '../flows' // separates is which of the two places one is kept in, which is the one thing that shows -- // the package is there before anything has been fetched, and the repository is not. const WHERE: { id: Place | 'all'; said: string; note: string }[] = [ - { id: 'all', said: 'every flow', note: 'eleven, and humanize 1 is three of them' }, + { id: 'all', said: 'every flow', note: 'fourteen, and humanize 1 is three of them' }, { id: 'package', said: 'in the package', note: 'chat, which is there before anything is fetched' }, - { id: 'flowverse', said: 'the official flowverse', note: 'humanfia/flowverse, fetched the first time somebody wants it' }, + { id: 'flowverse', said: 'the official flowverse', note: 'humanfia/flowverse, fetched as /flow first opens' }, ] const place = ref('all') diff --git a/docs/.vitepress/theme/components/HmzSurfaces.vue b/docs/.vitepress/theme/components/HmzSurfaces.vue index bb49d780..e66f1c19 100644 --- a/docs/.vitepress/theme/components/HmzSurfaces.vue +++ b/docs/.vitepress/theme/components/HmzSurfaces.vue @@ -9,6 +9,9 @@ type PlanMode = 'discussion' | 'direct' type SurfaceKey = 'python' | 'cli' | 'tui' | 'daemon' const PLAN_MODES: PlanMode[] = ['discussion', 'direct'] +// What the params form offers for `turn_retries`: one more than the model takes, so that the +// refusal the model answers with is one a reader can reach. +const RETRIES = [0, 1, 2, 3, 4] interface LookupRow { key: string @@ -39,8 +42,8 @@ interface GraphEdge { } const forked = ref(false) -const genIdea = ref(true) -const genPlan = ref(true) +const autoStart = ref(false) +const retries = ref(1) const planMode = ref('discussion') const surface = ref('tui') const attached = ref(1) @@ -76,16 +79,16 @@ const lookup = computed(() => ], ) -const configAccepted = computed(() => !genIdea.value || genPlan.value) +const configAccepted = computed(() => retries.value <= 3) const configState = computed(() => configAccepted.value ? [ 'accepted', - `idea ${genIdea.value ? 'on' : 'off'}`, - `plan ${genPlan.value ? 'on' : 'off'}`, planMode.value, + `retries ${retries.value}`, + `auto-start ${autoStart.value ? 'on' : 'off'}`, ].join(' · ') - : 'refused · gen idea is on while gen plan is off', + : 'refused · turn_retries: input should be less than or equal to 3', ) const resolved = computed(() => forked.value @@ -98,8 +101,9 @@ const SURFACES: Surface[] = [ key: 'python', label: 'Python SDK', about: - 'Builds a Run around the loaded runner and task, then runs it here or on a ' + - 'thread of its own.', + 'Hmz().run(flow, task, agents={role: spec}, envs=, params=, budget=, resume=, ' + + 'outworlder=) loads a Runner and hands back a Run of it and the task: run here or ' + + 'on a thread of its own, watched, stopped or closed.', path: [ 'python-workspace', 'workspace-run', @@ -113,8 +117,8 @@ const SURFACES: Surface[] = [ key: 'cli', label: 'CLI', about: - 'Reads the line into the same flow, agents, task and setup, then drives the SDK ' + - 'Run to its return.', + 'Hmz.read(argv) reads the line into a Line -- the flow, -a/-e/-p/-b and the task -- ' + + 'then drives the same Run to its return.', path: [ 'cli-workspace', 'workspace-run', @@ -128,8 +132,8 @@ const SURFACES: Surface[] = [ key: 'tui', label: 'TUI', about: - 'Keeps the workspace and runner in hand so it can configure, watch and steer the ' + - 'agent conversations while they run.', + 'Sets the flow up by role, keeps its Runner and Run in hand, and answers for the ' + + 'outworlder while it watches and steers the conversations.', path: ['tui-workspace', 'workspace-runner', 'runner-conversations', 'runner-epic'], nodes: ['tui', 'workspace', 'runner', 'conversations', 'epic'], }, @@ -190,8 +194,8 @@ function forkFlow() { function reset() { forked.value = false - genIdea.value = true - genPlan.value = true + autoStart.value = false + retries.value = 1 planMode.value = 'discussion' surface.value = 'tui' attached.value = 1 @@ -258,30 +262,37 @@ function reset() { -
gen-idea
- -
gen-plan
+
+
+ turn retries + how many times a failed or empty turn is retried +
+
+ +
+
+
plan mode @@ -310,9 +321,9 @@ function reset() { {{ configState }}

- Turn gen plan off while gen idea remains on. The model refuses the relationship; - the interface only shows what it said. The whole set is validated again when the - current flow is loaded. + Ask for four retries. The model takes at most three and refuses it; the interface + only shows what it said. The whole set is validated again when the current flow is + loaded.

diff --git a/docs/.vitepress/theme/flows.ts b/docs/.vitepress/theme/flows.ts index 4a69aab9..4d7acc58 100644 --- a/docs/.vitepress/theme/flows.ts +++ b/docs/.vitepress/theme/flows.ts @@ -68,7 +68,7 @@ export const FLOWS: Flow[] = [ link: '/flows/ralph-loop', agents: 'agent', said: 'A fresh session every round, so nothing carries over but the repository.', - ends: 'the run’s budget', + ends: 'three empty rounds in a row, or the run’s budget', keeps: 'rounds', place: 'flowverse', family: 'fresh', @@ -79,7 +79,7 @@ export const FLOWS: Flow[] = [ link: '/flows/stateful-ralph', agents: 'agent', said: 'One session, held for the whole run, re-sent the task every round.', - ends: 'the run’s budget', + ends: 'three empty rounds in a row, or the run’s budget', keeps: 'rounds', place: 'flowverse', family: 'held', @@ -90,7 +90,7 @@ export const FLOWS: Flow[] = [ link: '/flows/continue-loop', agents: 'agent', said: 'Sends the task once, then keeps nudging “continue” at the session that heard it.', - ends: 'the run’s budget', + ends: 'the run’s budget, or three failed turns in a row', keeps: 'rounds', place: 'flowverse', family: 'nudge', @@ -112,7 +112,7 @@ export const FLOWS: Flow[] = [ link: '/flows/flame-chase', agents: 'first_chaser · second_chaser', said: 'Two agents take turns on the same task, each reading the repository rather than a history.', - ends: 'the run’s budget, which the two spend between them', + ends: 'the run’s budget, which the two spend between them, or three failed turns in a row', keeps: 'turn · rounds', place: 'flowverse', family: 'pair', @@ -123,7 +123,7 @@ export const FLOWS: Flow[] = [ link: '/flows/rlar', agents: 'actor · reviewer', said: 'The actor remembers and the reviewer must not. The review is the actor’s next prompt.', - ends: 'the reviewer agreeing the work is done', + ends: 'the reviewer agreeing the work is done, or the run’s budget', keeps: 'rounds · the review nobody acted on', place: 'flowverse', family: 'review', @@ -134,7 +134,7 @@ export const FLOWS: Flow[] = [ link: '/flows/humanize1', agents: '1, then 2, then 2 + you', said: 'PolyArch/humanize as three flows: an idea, a plan both sides converged on, and a build under review.', - ends: 'its max rounds, for the loop of the three', + ends: 'the reviewer saying the plan is complete, or its max rounds, for the loop of the three', keeps: 'the directory the loop is in, and its round', place: 'flowverse', family: 'phases', @@ -159,6 +159,46 @@ export const FLOWS: Flow[] = [ place: 'flowverse', family: 'lanes', }, + { + name: 'ralph_loop_agent_cleanup', + link: '/flows/agent-cleanup', + agents: 'agent · cleaner + you', + said: 'ralph_loop, with a cleaner that distills the workspace into one commit every few turns.', + ends: 'the run’s budget, or three empty turns in a row', + keeps: 'the turns, the epoch, the run’s own directory', + place: 'flowverse', + family: 'fresh', + }, + { + name: 'flame_chase_agent_cleanup', + link: '/flows/agent-cleanup', + agents: 'first_chaser · second_chaser · cleaner + you', + said: 'flame_chase, with the same cleaner between the two chasers.', + ends: 'the run’s budget, or three empty turns in a row', + keeps: 'the turns, the epoch, the run’s own directory', + place: 'flowverse', + family: 'pair', + }, + { + name: 'recursive_lean_prover', + link: '/flows/recursive-lean-prover', + agents: 'worker · reviewer', + said: 'A Lean theorem proved by recursive decomposition, each node built by humanize1’s phases in a worktree of its own.', + ends: 'the root theorem proved or refused, or the run’s budget', + keeps: 'the DAG, the accepted nodes, their worktrees and branches', + place: 'flowverse', + family: 'phases', + }, + { + name: 'aot', + link: '/flows/aot', + agents: 'writer · critic + you', + said: 'Writes a flow from a description, and lands it once it has loaded, run on fakes and been read.', + ends: 'the flow landed, or its repairs running out', + keeps: '', + place: 'flowverse', + family: 'review', + }, ] /* ------------------------------------------------------------------------------------------ @@ -484,7 +524,7 @@ export const SHAPES: Record = { lanes: [ { id: 'builder', name: 'the builder', note: 'one session, the loop' }, { id: 'reviewer', name: 'the reviewer', note: 'fresh, each round' }, - { id: 'you', name: 'you', note: 'asked once, never waited', tone: 6 }, + { id: 'you', name: 'you', note: 'asked only when you are there', tone: 6 }, ], steps: [ { @@ -494,7 +534,7 @@ export const SHAPES: Record = { label: 'have you read the plan?', session: 'none', tone: 'ask', - carry: A('answered, or not', 'b1'), + carry: A('answered — skipped when you are away', 'b1'), }, { id: 'b1', @@ -570,7 +610,7 @@ export const SHAPES: Record = { parallel_flame_chase_git_pr: { of: 'parallel_flame_chase_git_pr', lanes: [ - { id: 'co', name: 'the orchestrateor', note: 'plans once', tone: 6 }, + { id: 'co', name: 'the orchestrator', note: 'plans once', tone: 6 }, { id: 'l1a', name: 'lane 1 · a', note: 'a clone of its own', tone: 1 }, { id: 'l1b', name: 'lane 1 · b', note: 'a clone of its own', tone: 1 }, { id: 'l2a', name: 'lane 2 · a', note: 'a clone of its own', tone: 2 }, diff --git a/docs/contributing/architecture.md b/docs/contributing/architecture.md index df817eec..e1fb7756 100644 --- a/docs/contributing/architecture.md +++ b/docs/contributing/architecture.md @@ -11,7 +11,6 @@ src/hmz/ ├── __main__.py python -m hmz ├── coganchor/ everything humanize knows about driving a coding agent CLI ├── flows/ the whole of what a flow imports, and nothing else: types -├── _legacy_flows/ the flow API before this one, until every way in has moved off it ├── runtime/ what a run is: driving one, writing it down, reading it back — │ flowing/, which is everything humanize does to a flow, and │ doing/, which is the whole of that as one object @@ -44,7 +43,7 @@ for the anchor inside it, a program that ships to a target and could be lifted o | `runtime/flowing/` | Everything humanize does to a flow, and none of it a thing a flow names. **The engine**: defining a flow, calling one, the views a flow is handed, budgets, the resume journal, refs and the module cache. **The drivers** the engine runs over, behind one SPI: an agent driver per harness over `coganchor`, and an environment driver for this machine and for ssh hosts. The parsers `-a`, `-e`, `-p` and `-b` are read with, the in-memory fake kit a flow is tested on, and where flows come from, finding one by name and the skills a flow brings. | `run_flow`, `load_flow`, `define_flow`, `running`, `open_agent`, `open_env`, `local_env`, `parse_agents`, `parse_envs`, `parse_params`, `parse_budget`, `run_fake`, `found`, `find`, `fork`, `flowverses`, `brought` | | `coganchor/fallbacks.py` | The layer between an agent and its accounts: where a turn goes when the place taking it cannot take it at all, and how many times over it is taken again first. A step is written between two places — `CLI[@ACCOUNT]/MODEL` — rather than on the account, which `providers` already answers for. Names `backends` and nothing else. | `Falls`, `falls`, `points`, `retrying`, `tried`, `clear`, `chain`, `spec`, `reads`, `waits`, `POLICIES` | | `runtime/epic.py` | One run of one flow as a directory: the journal, the links to each session's log, and what a flow that can be picked up left behind. Written by `runner`, read by `tracing`, `cli` and `tui`. | `Epic`, `epics`, `read`, `opened`, `state`, `resumed` | -| `runtime/runner.py` | Handing a flow the drivers its roles are filled with, and running it under an epic. Also reads the `hmz exec` line, which the interface starts a flow from too. | `Runner`, `flow_and_agents`, `read_agent`, `set_up_from` | +| `runtime/runner.py` | Handing a flow the drivers its roles are filled with, and running it under an epic. Also reads the `hmz exec` line, which the interface starts a flow from too. | `Runner`, `Recorder`, `Refused`, `Line`, `read_line` | | `runtime/tracing/` | Reading the backends' logs back — and, for a profiled run, sampling the programs its agents start — and rendering both as one Chrome trace. | `collect`, `profile.Profiler` | | `runtime/doing/` | humanize as one object, and the front door `hmz.runtime` hands through. A workspace, what is remembered about it, the flows there are, the agents and accounts they run as, the runs already made and the run being made now. It composes the layers and restates none of them, and it reaches each of them from inside the call that needs it — which is what lets a caller name it without paying for the tracer. | `Hmz`, `Run` | | `tui/` | The terminal interface. It reaches the runtime through the daemon holding the run it is drawing. | `Humanize` | @@ -106,9 +105,7 @@ runtime/flowing/ ├── fakes.py in-memory drivers for every seam, to test a flow on ├── finding.py a flow by name, nearest first ├── skills.py the skills a flow brings, its own and the ones it named -├── verses.py where flows come from when they come from somewhere else -└── checking.py driving.py prophecy.py prophesying.py stepping.py proving.py - the previous flow API's machinery, over _legacy_flows, and going with it +└── verses.py where flows come from when they come from somewhere else ``` The line between those two is the point of them. A flow is somebody else's repository, so diff --git a/docs/flows/agent-cleanup.md b/docs/flows/agent-cleanup.md new file mode 100644 index 00000000..49efe6cb --- /dev/null +++ b/docs/flows/agent-cleanup.md @@ -0,0 +1,75 @@ +--- +pageClass: hmz-feature +--- + +# ralph_loop_agent_cleanup · flame_chase_agent_cleanup + +[`ralph_loop`](/flows/ralph-loop) and [`flame_chase`](/flows/flame-chase), with a third agent +that cleans the workspace every few turns: it distills the work, deletes what strayed, writes +down what is next, and the repository's history is replaced by one commit of what survived. Two +flows, one implementation, kept identical by design. + +```sh +hmz exec -f ralph_loop_agent_cleanup \ + -a agent=claude/claude-opus-5:high -a cleaner=claude/claude-opus-5:high \ + -p work_paths=src -b duration=12h,cost=100 "$(cat TASK.md)" +hmz exec -f flame_chase_agent_cleanup \ + -a first_chaser=claude/claude-opus-5:high -a second_chaser=codex/gpt-5.6-sol:high \ + -a cleaner=claude/claude-opus-5:high \ + -p work_paths=src -b duration=12h,cost=100 "$(cat TASK.md)" +``` + +## Roles + +| | | +| --- | --- | +| `agent` — or `first_chaser` and `second_chaser`, alternating | The coding turns, a fresh session each | +| `cleaner` | One cleaning epoch at a time, keeping one session through its repairs | +| `human` | You — the [outworlder](/features/human), filled in by the runtime and never by `-a` | + +Every agent role is declared with `SteeringAgentMixin`: a turn is steered to wrap up when its +clock runs out, so each takes a harness that can be told something mid-turn — Claude Code, +Codex, Kimi Code or pi — and any other is refused before the first turn. The workspace is the +directory the run was started in, worked in through its shell. + +## A turn, and an epoch + +Every coding turn is a fresh session. After `session_timeout_minutes` it is steered to wrap up, +and `stop_grace_minutes` later it is cut off; a reminder is steered in when it has gone +`idle_timeout_minutes` without spending anything. A turn that answered nothing, or whose harness +failed, is taken again on the same seat, and three of those in a row end the run. + +Every `cleanup_turns` turns, between turns, the tree is saved aside and a cleaner session +distills the `work_paths`, deletes strays outside them and writes `NEXT.md`. What survived is +measured, and handed back up to `repairs` times; `check_command`, where there is one, is run on +the result, and a failure restores the tree. Then the history is archived and replaced by one +commit, `epoch N: distilled tree`. An epoch that is stopped, out of budget or failed restores the +tree before the run ends. The replaced history is never deleted: every epoch's is kept in the +workspace's `history.git` under humanize's home, `~/.humanize///`. + +## What it takes + +`work_paths` is required — the paths, relative to the repository, where agents may create or +revise task work: `-p work_paths=src,include`, or a JSON list. The rest have defaults: + +| | | +| --- | --- | +| `cleanup_turns` | Coding turns between epochs; `3`, and `0` never cleans | +| `next_lines` · `comment_lines` | The most lines `NEXT.md` may hold, `10`, and the comment-line cap under the work paths, `30` | +| `repairs` | Over-measures handed back to the cleaner; `2` | +| `check_command` | A correctness check after cleaning; `""` skips it | +| `session_timeout_minutes` · `stop_grace_minutes` · `idle_timeout_minutes` | `240`, `10` and `20`; `0` disables either timeout | +| `max_tracked_file_mb` | Files larger than this, `10`, are never committed | +| `confirm_large_workspace_copies` | Ask before cleaning a workspace past 5,000 files or 1 GiB; `true` — and under `hmz exec`, where nobody answers, the run does not start | + +## What ends it + +The run's [budget](/features/allowances): whichever limit of `-b` is reached first stops it, and +an epoch it stops is put back first. Three turns in a row that came to nothing end it sooner. It +can be picked up with `--resume`, which carries on the run's turn count and epochs. + +## See also + +- [ralph_loop](/flows/ralph-loop) · [flame_chase](/flows/flame-chase) — the loops, without the + cleaning +- [A line typed mid-turn](/features/steering) — what steering a turn is diff --git a/docs/flows/aot.md b/docs/flows/aot.md new file mode 100644 index 00000000..f60c472a --- /dev/null +++ b/docs/flows/aot.md @@ -0,0 +1,66 @@ +--- +pageClass: hmz-feature +--- + +# aot + +The flow that writes a flow: a description in, and out a flow that has been loaded through +humanize's own engine, run on fakes, and read by a critic. It lands in `.humanize/flows/` +under its name, ready for `hmz exec -f`. + +```sh +hmz exec -f aot -a writer=claude/claude-opus-5:high -a critic=codex/gpt-5.6-sol:high \ + -b cost=10 "two agents take turns until a reviewer says it is done" +``` + +## Three roles + +| | | +| --- | --- | +| `writer` | Draws a spec from the description, then drafts the flow in a scratch directory and repairs it | +| `critic` | Reads each draft that passed the gates, in a fresh session; declared `Permission(local=READ)`, so it reads and never writes | +| `human` | You — the [outworlder](/features/human), filled in by the runtime and never by `-a` | + +Both agents carry the flow's own [skill](/user/skills), `writing-flows`: the flow API, as a +flow is written against it. Any harness can fill either. + +## A draft is run before it lands + +The writer draws a spec first: the roles the flow drives, what each must be able to do, its +params and what ends it. A capability nothing here serves is sent back for the writer to +restate, and then put to you to narrow. The name is settled before anything is drafted: one +already taken in the place it lands is put to you, twice at most. + +Each draft then goes through two gates, in a process of its own: + +1. It is **loaded** through the engine, and what it declares is read back. +2. It is **run on the fake kit** in three worlds — an agent that never says done, one that says + done at once, one that answers nothing — and must end on its own in every one, within + `seconds` apiece. + +A draft that passes is read by the critic. A refusal from either goes back to the writer word +for word, for up to `repairs` rounds. + +## What it takes + +| | | +| --- | --- | +| `name` | What to call the flow that lands; `""` takes the name the spec derives from the description | +| `into` | `local`, this project's `.humanize/flows/`, or `user`, the one in your home directory | +| `repairs` | Rounds of repair after the first draft; `3`, from `0` to `6` | +| `strict` | Whether every warning sends a draft back rather than only what blocks; `false` | +| `seconds` | The clock each smoke run is held to; `60` | + +## What ends it + +The flow that lands, or none. What passes is moved into place whole, and never over a flow that +is already there. Under `hmz exec` nobody is at the prompt, so every question put to you is +answered no, and a draft that runs out of repairs is not landed. It prints the line that runs +what it wrote. + +It keeps nothing: a run of it is one description, and running it again writes another flow. + +## See also + +- [Writing a flow](/weaver/writing-a-flow) — the same thing, by hand +- [Testing a flow](/weaver/testing-flows) — the fake kit the gates run a draft on diff --git a/docs/flows/chat.md b/docs/flows/chat.md index 9f79eb73..fa0e2a57 100644 --- a/docs/flows/chat.md +++ b/docs/flows/chat.md @@ -20,8 +20,9 @@ runs under `Budget(cost=inf)`. ## Two agents, and the second is you `chat` drives an `assistant` and a `human` — the [outworlder](/features/human), whoever is outside -the run, filled in by the runtime and never by `-a`. Saying something to the person is asking -what to say next, and what they answer with is what they typed: +the run, filled in by the runtime and never by `-a` — in `workspace`, a `LocalEnv`: the directory +it was started in, which no `-e` names either. Saying something to the person is asking what to +say next, and what they answer with is what they typed: ```python conversation = await assistant.spawn(env=workspace) @@ -43,7 +44,8 @@ is said to you, and by then there is a conversation to carry on. ## Every capability the harness has -`chat` declares a plain `Agent`, because it talks to whichever harness it is given — and it is +`chat` declares a plain `Agent` — allowed the web, `Permission(online=PermissionKind.ALL)`, as a +person talking to one would expect — because it talks to whichever harness it is given; and it is the one flow the runtime hands the harness's **full** view, so everything that harness can do is there: a `/goal` typed at it on a harness with one, a question the agent stops to ask put to you on a harness that asks. Every other flow gets exactly what it declared. diff --git a/docs/flows/continue-loop.md b/docs/flows/continue-loop.md index e12feb54..c70aaaaa 100644 --- a/docs/flows/continue-loop.md +++ b/docs/flows/continue-loop.md @@ -24,19 +24,28 @@ sends the task, and only once a turn has actually landed does the prompt become try: answered = await agent.run(prompt, session=session) except HarnessError: + failed += 1 + if failed >= 3: + raise answered = "" +else: + failed = 0 if answered: prompt = "continue" ``` -A turn that failed — a backend that fell over before it said anything — lands nothing, and the -next round sends the task rather than nudging a session that never got it. +A turn that failed or answered nothing — a backend that fell over before it said anything — is +sent again: the task until a turn has answered, `continue` after. Three failed turns in a row +end the run with the last failure. ## What ends it The run's [budget](/features/allowances) — `-b duration=…,cost=…,output_tokens=…` — held to at every turn of the session rather than implemented here. The flow itself takes no params and -declares no budget of its own, so `hmz exec` refuses to start it without a `-b`. +declares no budget of its own, so `hmz exec` refuses to start it without a `-b`. The turn that +finds it spent raises `BudgetExceeded`, which is how the run ends, and `--resume` carries on +counting rounds under a fresh `-b`. Three failed turns in a row end it sooner, with the last +failure. ## What it keeps diff --git a/docs/flows/flame-chase.md b/docs/flows/flame-chase.md index ba1d769c..584a8d5b 100644 --- a/docs/flows/flame-chase.md +++ b/docs/flows/flame-chase.md @@ -31,7 +31,11 @@ The run's [budget](/features/allowances) — `-b duration=…,cost=…,output_to spend it between them** rather than apiece, and that is the ordinary case rather than this flow's own arithmetic: a budget is the run's, and every agent of a run spends out of the one reckoning whichever of them was writing. The flow itself takes no params and declares no budget -of its own, so `hmz exec` refuses to start it without a `-b`. +of its own, so `hmz exec` refuses to start it without a `-b`. A spent budget raises +`BudgetExceeded`, and `--resume` starts with whichever chaser was next. + +A turn that fails passes to the other chaser; three failures in a row end the run with the last +one. ## What it keeps diff --git a/docs/flows/humanize1.md b/docs/flows/humanize1.md index 14953d61..6752b9e9 100644 --- a/docs/flows/humanize1.md +++ b/docs/flows/humanize1.md @@ -20,8 +20,10 @@ hmz exec -f humanize1:rlcr \ -b duration=2d,cost=300 -p max=20 "build it" ``` -`rlcr`'s third role, `human`, is you — the [outworlder](/features/human), filled in by the -runtime rather than by `-a`. +`humanize1` has no flow of the directory's own name, so a bare `-f humanize1` is refused, +listing the three: name a phase, `humanize1:`. All three work in `workspace`, the +directory the run was started in, and `rlcr`'s third role, `human`, is you — the +[outworlder](/features/human), filled in by the runtime rather than by `-a`. Three rather than one because each is set up on its own agents, and what passes between them is a file, as it is in the plugin — the draft, then the plan. So an idea may be opened on one @@ -36,16 +38,22 @@ review reads what came after it. One agent, `n` directions explored at once, one draft written out. `n` and `output` are the two -params, under the names the plugin gives them: `-p n=5,output=IDEA.md`. +params, under the names the plugin gives them: `-p n=5,output=IDEA.md`. `n` is 6 unless it is +given, from 2 to 10; an `output` left blank writes `.humanize/ideas/-.md`, and one +that already exists is refused. ## 2 · `gen-plan` The planner holds one session for the whole of the planning; the analyst arrives fresh each -time and reads the plan against the repository. They converge, or the round limit stops them. -`input` names the draft to plan from, `mode` is `discussion` or `direct`, and -`alternative_plan_language` writes a translated plan beside the plan. +time and reads the plan against the repository. They converge — up to three review rounds, +stopping after two in which nothing material changed. `input` names the draft to plan from (the +newest `.humanize/ideas/*.md` where it is blank), `output` the plan to write (`docs/plan.md`, +which must not exist yet), `mode` is `discussion` or `direct`, `alternative_plan_language` +writes a translated plan beside the plan, and `turn_timeout`, `total_timeout` (3600 and 14400 +seconds, `0` for none) and `turn_retries` (1) bound the agents' turns. A decision the two left +`PENDING` fails the run once the plan is written. ## 3 · `rlcr` @@ -58,11 +66,14 @@ here it is the flow's own loop, builder, gates and reviewer, which reads the sam harness. The plugin's tool validators are hooks, as they are there: an [`on_permission_request` hook](/features/hooks) on the builder, which is why the builder's role is declared with `PermissionRequestHookAgentMixin` and has to be a harness that asks — Claude -Code, Codex, Kimi Code or ZCode. +Code, Codex, Kimi Code or ZCode — beside an `on_pre_tool_use` hook that watches what it reads +and its task list, and an `on_user_prompt_submit` hook on its opening prompt. Every flag the plugin takes is a field on that phase's own params, under the plugin's own name -for it: `max`, `full_review_round`, `skip_code_review`, `plan_file`, `base_branch`, and the rest. -`-p` sets any of them, and the params form in `/flow` is all of them. +for it: `max` (42), `full_review_round` (5), `codex_timeout` (5400 seconds), `skip_code_review`, +`plan_file` (`docs/plan.md` where it is blank), `base_branch` (`origin/HEAD`, then `main`, then +`master` where it is blank), `skip_quiz`, `yolo`, and the rest. `-p` sets any of them, and the +params form in `/flow` is all of them. The task on the line is not read: the plan is the task. It writes what the plugin writes, where the plugin writes it: `.humanize/rlcr//` with `state.md`, `goal-tracker.md`, and a prompt, summary, contract and review per round — so @@ -77,13 +88,16 @@ The plugin's mechanism, where humanize's is not the same mechanism: | `codex review --base ` | Takes no prompt and is a Codex feature. Here the reviewer is whichever agent was chosen, so the code review is **asked for**, in a prompt that asks for exactly the `[P0-9]` output the loop then reads the same way. | | `--codex-timeout` | A review that runs past it is treated as a review that failed, which is the state the plugin's own timeout leaves the round in. | | `/humanize:ask-codex` | A task the plan tags `analyze` is a shell script the builder runs there. Here the builder has no way to reach the reviewer mid-round, so it is told to put the question in its round summary, where the reviewer answers it. | -| The plan quiz | Put to `human`, the outworlder, the way a coding agent's own question is put — so a run with nobody at the prompt is answered for at once rather than waiting for a person who is not there. | +| The plan quiz | Put to `human`, the outworlder, only when somebody is there: the reviewer writes the questions and you pick an answer to each. With nobody at the prompt — `hmz exec`, `/afk` — it is skipped, and no reviewer turn is spent on it. | ## What it keeps `rlcr` is meant to run for days, so a run of it can be picked up with `--resume`: it keeps -**which** `.humanize/rlcr/` directory the loop is in and the round it reached, and reads -`state.md` back as it stands rather than stamping a new directory beside a week of rounds. Everything else is +**which** `.humanize/rlcr/` directory the loop is in and the round it reached, reads `state.md` +back as it stands rather than stamping a new directory beside a week of rounds, and sends the +round's saved prompt to a new builder session. It ends `complete` — the reviewer saying so and a +clean code review — `maxiter` on its `max` rounds, `stop` where the reviewer or the drift breaker +says to, or `unexpected`. Everything else is already in that directory in the plugin's own format, and a second copy here would be a second place for it to be wrong. diff --git a/docs/flows/index.md b/docs/flows/index.md index d77a5550..8dfcecef 100644 --- a/docs/flows/index.md +++ b/docs/flows/index.md @@ -9,8 +9,8 @@ each is asked, in what order, and when to stop. humanize runs flows and has no o what a good one is — so a flow is content rather than product, whoever writes one is a **weaver**, and the list below is something to read, fork, publish and beat. -Ten are drawn here — twelve by name, since [`humanize1`](/flows/humanize1) is three phases. -Between them they are most of the loop shapes the field has converged on. +Fourteen are listed here — sixteen by name, since [`humanize1`](/flows/humanize1) is three +phases. Between them they are most of the loop shapes the field has converged on. @@ -34,6 +34,9 @@ honest answers. | A plan agreed first, then built under review | [`humanize1`](/flows/humanize1) | | Three streams of work at once, only one of them touching your tree | [`parallel_flame_chase`](/flows/parallel-flame-chase) | | Three lanes, each with a clone, merged into `main` only by a measurement | [`parallel_flame_chase_git_pr`](/flows/parallel-flame-chase-git-pr) | +| A long loop whose workspace is distilled every few turns | [`ralph_loop_agent_cleanup`, `flame_chase_agent_cleanup`](/flows/agent-cleanup) | +| A Lean theorem proved by recursive decomposition | [`recursive_lean_prover`](/flows/recursive-lean-prover) | +| A flow written for you from a description | [`aot`](/flows/aot) | Six name a [FlowBench](https://humanfia.ai/projects/flowbench) loop in their own docstring, so that comparing one method against another is a flag rather than a reimplementation. @@ -81,15 +84,20 @@ is what makes a run stopped by its budget a run to pick up rather than one that Some reach an end of their own first: [`chat`](/flows/chat) when you stop typing — the one flow that runs without a `-b` — [`rlar`](/flows/rlar) when its reviewer agrees the work is done, [`goal`](/flows/goal) when the model says the objective is met, and -[`humanize1`](/flows/humanize1)'s loop on its `max` rounds. For the two +[`humanize1`](/flows/humanize1)'s loop when its reviewer says the plan is complete, or on its +`max` rounds. The loops of one agent stop after three rounds in a row that came to nothing, and +several end with the error after three failed turns in a row. For the two [lane flows](/flows/parallel-flame-chase) the budget is the only end there is: their lanes are scheduled again for as long as they run, so give them a duration. +A run its budget stopped ends with `BudgetExceeded` — `hmz exec` says which limit and exits 0 — +and one that can be picked up carries on from there with `--resume` and a fresh `-b`. + ## Where they come from | | | | --- | --- | -| `official` | humanize's own, which is [`chat`](/flows/chat) in the package and [humanfia/flowverse](https://github.com/humanfia/flowverse) for everything else, fetched the first time somebody wants what is in it | +| `official` | humanize's own, which is [`chat`](/flows/chat) in the package and [humanfia/flowverse](https://github.com/humanfia/flowverse) for everything else, fetched as `/flow` first opens or with `r` at `/flowverses` — until then `hmz exec` naming one of its flows says so | | `local` · `user` | `.humanize/flows/` here, and `~/.humanize/flows/` everywhere | Which of humanize's two places a flow is kept in is humanize's business, so all of them are diff --git a/docs/flows/parallel-flame-chase-git-pr.md b/docs/flows/parallel-flame-chase-git-pr.md index 4e9fe68e..09563ebb 100644 --- a/docs/flows/parallel-flame-chase-git-pr.md +++ b/docs/flows/parallel-flame-chase-git-pr.md @@ -11,7 +11,7 @@ configuration that did best in a twelve-hour Git PR Lite experiment, and nothing ```sh hmz exec -f parallel_flame_chase_git_pr \ - -a orchestrateor=codex/gpt-5.6-sol:max \ + -a orchestrator=codex/gpt-5.6-sol:max \ -a lane_1_actor_a=codex/gpt-5.6-sol:max,lane_1_actor_b=claude/claude-opus-5:max \ -a lane_2_actor_a=claude/claude-opus-5:max,lane_2_actor_b=codex/gpt-5.6-sol:max \ -a lane_3_actor_a=claude/claude-opus-5:max,lane_3_actor_b=codex/gpt-5.6-sol:max \ @@ -24,13 +24,16 @@ hmz exec -f parallel_flame_chase_git_pr \ | | | | --- | --- | -| `orchestrateor` | Plans the three lanes, once | +| `orchestrator` | Plans the three lanes, once | | `lane_1_actor_a` · `lane_1_actor_b` | Lane 1, alternating in fresh turns, in a clone of its own | | `lane_2_actor_a` · `lane_2_actor_b` | Lane 2, the same | | `lane_3_actor_a` · `lane_3_actor_b` | Lane 3, the same | The eighth role, `human`, is you — the [outworlder](/features/human), filled in by the runtime and never by `-a`. There is no reviewer role: nothing here asks a model whether a change is good. +The lane actors declare `Permission(user=ALL)` — they push to the run's central repository and +write receipts outside their own clone — every role carries the flow's skill, and none is +declared with the goal mixin, so any harness can fill any of them. ## A pull request is merged by a measurement @@ -53,22 +56,29 @@ turn the measured workflow into a different one by accident. ## What it takes -The base flow's params — `rest_seconds`, `resume_mode`, the large-workspace thresholds — each a -`-p`. Before it makes its copies it measures the source, and warns where the planning tree, the -three lane clones, the integration clone and the git objects between them come to more than the -thresholds; `-p confirm_large_workspace_copies=true` makes it ask first instead. +The base flow's params — `rest_seconds` (`1.0`), `resume_mode` (`auto`), the large-workspace +thresholds (`5000` files, `1073741824` bytes) — each a `-p`. Before it makes its copies it +measures the source, and warns where the planning tree, the three lane clones, the integration +clone and the git objects between them come to more than the thresholds; +`-p confirm_large_workspace_copies=true` makes it ask first instead — and under `hmz exec`, +where nobody is there to answer, the run does not start. The [skill](/user/skills) it brings, `parallel-flame-chase-git-pr`, is the lane, pull-request and receipt protocol, carried by every session the flow opens. Like the base flow, its lanes go on for as long as it runs, so the run's -[budget](/features/allowances) is its end: give `-b` a duration. +[budget](/features/allowances) is its end: give `-b` a duration. When it is spent the turns under +way land and are recorded, and `BudgetExceeded` ends the run — `--resume` carries it on. ## What it keeps The central repository's refs, the receipts, the artifacts, the report archive and the official -ledger, for a run picked up with `--resume`. The original source is assumed not to change -outside the flow while it holds the source lock. +ledger, for a run picked up with `--resume` — all of it in a scratch directory of the workspace +under humanize's home, `~/.humanize/envs/-/scratch/parallel_flame_chase_git_pr--…`, +which a call from a flow that cannot be picked up has removed when it ends. `-p resume_mode=fresh` +starts another run. The original source is assumed not to change outside the flow while it holds +the source lock, which it shares with [`parallel_flame_chase`](/flows/parallel-flame-chase): a +second run over the same source, of either flow, refuses to start. ## See also diff --git a/docs/flows/parallel-flame-chase.md b/docs/flows/parallel-flame-chase.md index a2f06641..e7ec597c 100644 --- a/docs/flows/parallel-flame-chase.md +++ b/docs/flows/parallel-flame-chase.md @@ -33,21 +33,26 @@ never by `-a`: | `lane_2_actor_a` · `lane_2_actor_b` | Lane 2, alternating, in a snapshot of its own | | `lane_3_actor_a` · `lane_3_actor_b` | Lane 3, alternating, in a snapshot of its own | -None of the seven is declared with the [goal](/features/goals) mixin, because a lane's turn -ends where the lane protocol says it ends rather than where a model decides it has met the -objective. The flow declares it, so it holds for whichever agents the run is given: a `/goal` -through any of them is refused. +Every one of the seven declares full permission — `Permission(local=ALL, user=ALL, system=ALL, +online=ALL)` — and the flow's skill, so any harness can fill any of them. None is declared with +the [goal](/features/goals) mixin, because a lane's turn ends where the lane protocol says it +ends rather than where a model decides it has met the objective. The flow declares it, so it +holds for whichever agents the run is given: a `/goal` through any of them is refused. ## One writer, and two that cannot write -A per-source advisory lock permits only one lane 1 owner; lanes 2 and 3 are confined to +A per-source lock permits only one lane 1 owner — shared with +[`parallel_flame_chase_git_pr`](/flows/parallel-flame-chase-git-pr), so a run of either refuses a +source the other holds; lanes 2 and 3 are confined to snapshots, and the runtime's control paths reject links and replacements. What they produce reaches lane 1 as a **report** and a hashed, reconstructable artifact package, and reports are redelivered until the receiving lane completes a valid turn and acknowledges them, so a lane that fell over does not lose what it was told. -Durable data lives under `~/.humanize/parallel_flame_chase///`. The flow -coordinates local work only: there is no release, deployment, submission, messaging or purchase +Durable data lives in a scratch directory of the workspace, +`~/.humanize/envs/-/scratch/parallel_flame_chase--/`: kept +for `--resume` by a run that can be picked up, and removed when the flow is called from one that +cannot. Lane 1's work in the source stays either way. The flow coordinates local work only: there is no release, deployment, submission, messaging or purchase executor in it. ## What it takes @@ -56,16 +61,18 @@ Its params, each a `-p`: | | | | --- | --- | -| `rest_seconds` | what the single-writer scheduler rests between control passes; `1.0` | +| `rest_seconds` | what the single-writer scheduler rests between control passes; `1.0`, from `0.05` to `60` | | `resume_mode` | `auto`, or `fresh` to deliberately start another run | -| `confirm_large_workspace_copies` | whether to ask before copying a very large workspace; `false` | -| `workspace_file_warning_threshold` · `workspace_copy_warning_threshold_bytes` | what counts as very large | +| `confirm_large_workspace_copies` | whether to ask before copying a very large workspace; `false`. With `true`, a run with nobody to answer — `hmz exec` — stops before copying | +| `workspace_file_warning_threshold` · `workspace_copy_warning_threshold_bytes` | what counts as very large; `5000` files, `1073741824` bytes | The [skill](/user/skills) it brings, `parallel-flame-chase`, is the actor, report, artifact, checkpoint and resume protocol — carried by every session the flow opens. Its lanes are scheduled again for as long as it runs, so the run's -[budget](/features/allowances) is the only end there is: give `-b` a duration. +[budget](/features/allowances) is the only end there is: give `-b` a duration. When it is spent +the turns under way land and are recorded, the run is marked stopped, and `BudgetExceeded` +ends it — `--resume` carries it on under a fresh `-b`. ## What it keeps diff --git a/docs/flows/recursive-lean-prover.md b/docs/flows/recursive-lean-prover.md new file mode 100644 index 00000000..30583cf4 --- /dev/null +++ b/docs/flows/recursive-lean-prover.md @@ -0,0 +1,69 @@ +--- +pageClass: hmz-feature +--- + +# recursive_lean_prover + +A theorem proved in Lean by recursive decomposition: each node is planned, proved in prose, +split into child theorems where it has to be, and formalized in a git worktree of its own — +every accepted theorem written to a Markdown wiki as it lands. It is built out of +[`humanize1`](/flows/humanize1)'s phases, called in the same process. + +```sh +hmz exec -f recursive_lean_prover \ + -a worker=codex/gpt-5.6-sol:max -a reviewer=codex/gpt-5.6-sol:max \ + -p lean_target=Submission.lean -p 'comparator_command=bash tools/check-with-comparator.sh' \ + -b duration=72h "$(cat PROBLEM.md)" +``` + +Run it at the root of a clean Lean git repository that ignores `.humanize/`, with a comparator +wrapper that exits zero and prints `comparator_success` only when every check has passed. + +## Two roles + +| | | +| --- | --- | +| `worker` | Writes each plan, proof and Lean formalization; declared with `PermissionRequestHookAgentMixin`, since `humanize1:rlcr`'s guards work through its permission requests — so Claude Code, Codex, Kimi Code or ZCode | +| `reviewer` | Checks each proof and each candidate afresh, and reruns the comparator itself | + +Both declare `Permission(local=ALL, user=ALL, system=READ, online=ALL)` and carry the flow's +skill, `recursive-lean-proof`. Every turn is a fresh session. There is no `human` role. + +## Each node + +1. **One scaffold plan**, from `humanize1:gen-plan` in `direct` mode, handed `planner=worker` + and `analyst=reviewer`, and never regenerated. +2. **A proof in prose**, revised from the reviewer's first invalid step, in batches of + `natural_proof_attempts`. +3. **A decomposition**, where the node needs one: child theorems with exact Lean statements and + an acyclic dependency list, each child going through the same steps — up to `max_depth` deep, + `max_children` apiece, `max_nodes` in all. +4. **Formalization in a worktree of its own**: every ready node gets a named branch and a + `derive_worktree` checkout under a scratch directory of the workspace, and runs + `humanize1:rlcr` through the hidden `worktree-rlcr` subflow with that worktree as its + workspace, for up to `rlcr_rounds` rounds. Up to `max_parallel_children` nodes work at once. +5. **Acceptance**: the comparator, then a fresh reviewer that reruns it. An accepted theorem is + published to the wiki and unlocks what depends on it. + +## What it takes + +Every param has a default: `max_depth` `2`, `max_children` `4`, `max_parallel_children` `24`, +`max_nodes` `24`, `natural_proof_attempts` `3`, `decomposition_attempts` `2`, `rlcr_rounds` +`20`, `plan_turn_timeout` `3600` and `plan_total_timeout` `14400` seconds, `comparator_timeout` +`21600` seconds, `lean_target` blank for the worker to infer, `comparator_command` +`bash tools/check-with-comparator.sh`, `comparator_success` `Your solution is okay!`, +`artifact_dir` `.humanize/recursive-lean-prover`, `wiki_dir` `.humanize/math-wiki`, and +`stop_on_child_failure` `true`. + +## What ends it + +The root theorem proved, or not accepted — said, with why, and kept for the next run — or the +run's [budget](/features/allowances) spent. A refused credential or a model that is not served +stops it too. It can be picked up with `--resume`, in the same repository: the run, its accepted +nodes, their worktrees and branches, the wiki and the latest rejected draft are all reused. + +## See also + +- [humanize1](/flows/humanize1) — the phases it is built from +- [A flow that calls a flow](/weaver/calling-flows) +- [Worktrees](/weaver/worktrees) — what `derive_worktree` makes diff --git a/docs/flows/rlar.md b/docs/flows/rlar.md index a0c922ea..da4127c9 100644 --- a/docs/flows/rlar.md +++ b/docs/flows/rlar.md @@ -28,16 +28,18 @@ class Review(BaseModel): notes: str # the review itself, written as a message to the coding agent ``` -`notes` becomes the actor's next prompt verbatim. `done` is what ends the run — this is the one -flow here that ends on a judgement rather than on running out. The run's -[budget](/features/allowances) is under it as it is under every flow, and it is the ceiling -rather than the point: what ordinarily stops this one is the reviewer agreeing. +`notes` becomes the actor's next prompt verbatim. `done` is what ends the run — it ends on a +judgement rather than on running out, as [`goal`](/flows/goal) and +[`humanize1`](/flows/humanize1)'s loop do. The run's [budget](/features/allowances) is under it +as it is under every flow, and it is the ceiling rather than the point: what ordinarily stops +this one is the reviewer agreeing. A turn that fails, or a review out of shape, is taken again +the next round; three in a row end the run with the last failure. The reviewer's prompt tells it to be skeptical, and to treat reward hacking — tests weakened or special-cased, work stubbed out or faked — as the thing it is most there to catch. How to read a round of work, and how to write the review the actor is then handed, is the flow's own -[skill](/user/skills): `skills/review-notes`, which its roles name and every session of theirs -carries. A weaver who wants the reviews written differently forks the flow, edits that one file, +[skill](/user/skills): `skills/review-notes`, which the `reviewer` role names, so every review +session carries it — the actor's sessions do not. A weaver who wants the reviews written differently forks the flow, edits that one file, and runs. ## Give the two the same model, if you like diff --git a/docs/reference/agents.md b/docs/reference/agents.md index 7d4fa02c..8ae2a30c 100644 --- a/docs/reference/agents.md +++ b/docs/reference/agents.md @@ -1979,7 +1979,7 @@ against a list fetched from [OpenLLMPrices](https://openllmprices.com/) and kept `~/.humanize/prices.json`: ```python -from hmz import prices +from hmz.coganchor import prices prices.cost(agent.spent(), agent.config.model) # dollars, or None for an unlisted model prices.price("claude-haiku-4-5-20251001") # Price(model="claude-haiku-4.5", …) diff --git a/docs/reference/cli.md b/docs/reference/cli.md index e8934a3e..43e7e7df 100644 --- a/docs/reference/cli.md +++ b/docs/reference/cli.md @@ -263,13 +263,23 @@ already behind it: required role left unfilled — a role the flow declares `NotRequired` may be left out; - an agent whose harness cannot serve what its role declares, or a role typed as one harness's own protocol given another harness; -- an environment short of what its role declares; +- an environment short of what its role declares, or one that cannot be reached — every + environment the line gives is probed before the flow is called; - params the flow's model refuses; - a skill a role names that cannot be found or fetched. +Each is one line on stderr, `hmz exec: error: …`, and exit status 2. A line argparse itself +cannot read — an unknown flag, no `-f` or task, an `-a`, `-e`, `-p` or `-b` that is not one — +prints argparse's usage line first: + ```console $ hmz exec -f rlar -a actor=claude/claude-opus-5:high -b cost=20 "fix the build" -hmz exec: error: rlar:rlar: no agent was given for 'reviewer' +hmz exec: error: rlar needs an agent for 'reviewer'; give each with -a ROLE=CLI/MODEL:EFFORT +$ hmz exec -f rlar -a actor=opus -b cost=20 "fix the build" +usage: hmz exec [-h] -f FLOW [-a ROLE=SPEC[,...]] [-e ROLE=SPEC[,...]] + [-p KEY=VALUE[,...]] [-b KEY=VALUE[,...]] [--resume] [--json] + task +hmz exec: error: -a 'actor=opus': expected [NAME=]CLI[@PROVIDER]/MODEL:EFFORT ``` Whatever else a flow does as it is imported is the flow's own, and fails as it would anywhere. @@ -277,13 +287,22 @@ Whatever else a flow does as it is imported is the flow's own, and fails as it w ### Picking a run up A flow that says it [can be picked up](/reference/flows#a-flow-that-can-be-picked-up) keeps a -journal of what it did while it runs. `--resume` carries on the newest such run of the same flow -in this workspace: the flow at the top picks up what it kept, and each flow it calls picks up -where it is called again with the same task, agents, environments and params. The line still -says what to run it on — its own `-a`, `-e`, `-p` and a fresh `-b` — so an agent that changed -is a flow called afresh from there down. - -Without `--resume` every run starts afresh, whatever an earlier one left behind. +journal of what it did while it runs: `resume.jsonl`, inside the run's +[epic](/user/tracing#what-a-run-writes-down), each state write a `{"t":"set",…}` line of it. `--resume` carries on the +newest run of the same flow in this workspace that got as far as writing one: the flow at the +top picks up what it kept, and each flow it calls picks up where it is called again with the +same task, agents, environments and params. The line still says what to run it on — its own +`-a`, `-e`, `-p` and a fresh `-b` — so an agent that changed is a flow called afresh from there +down. The run picking one up is an epic of its own, handed a copy of that journal, and says +which epic it `picked_up` from. + +A flow that is not resumable, or one with no such run here, is refused with `--resume`. Without +it every run starts afresh, whatever an earlier one left behind. + +A loop that only its budget ends ends that way: the turn it was in is let finish (unless the +budget said `graceful=false`), the run stops, `hmz exec: stopped -- …` says which limit it +reached, and the exit status is 0. A resumable one carries on from there with `--resume` and a +fresh `-b`. ### Examples @@ -590,9 +609,9 @@ into them. | | | | --- | --- | -| `0` | It did what it was asked. | +| `0` | It did what it was asked — a run its budget stopped included. | | `1` | It could not: the target could not be reached, the listener could not be started, a turn could not be supervised. | -| `2` | The command line was wrong — argparse's own rejections, a flow that is not there, a `-a`, `-e`, `-p` or `-b` that cannot be read or does not meet what the flow declares, no `-b` for a flow that is not `chat`, a malformed listen address, a non-loopback listener with no token. | +| `2` | The command line was wrong — argparse's own rejections, a flow that is not there, a `-a`, `-e`, `-p` or `-b` that cannot be read or does not meet what the flow declares, no `-b` for a flow that is not `chat`, an environment that cannot be reached, `--resume` with nothing to pick up, a malformed listen address, a non-loopback listener with no token. | | `130` | Interrupted. | | *the agent's own* | `hmz internal anchor` exits with the status of the program it ran, and `hmz internal cred` with that of the program it supervised. | diff --git a/docs/reference/flows.md b/docs/reference/flows.md index 35a369f1..1ffd3c19 100644 --- a/docs/reference/flows.md +++ b/docs/reference/flows.md @@ -916,6 +916,8 @@ of them — and two of one name are refused with `FlowDefinitionError`. `hidden=True` keeps an implementation flow, used only by the flows that load it, out of the lists and the `/flow` picker without losing its name: it remains callable as `:`. +The lists show the flow a bare name means under the directory's name, and every other visible +flow of the module as `:`. ## A flow that calls another flow @@ -1202,7 +1204,7 @@ is the conversation, and the harness logged it. ## The official flowverse Everything else humanize offers is in [humanfia/flowverse](https://github.com/humanfia/flowverse), -fetched the first time somebody wants what is in it. [Flows](/flows/) is the same list with the +fetched as `/flow` first opens, or with `r` at `/flowverses`. [Flows](/flows/) is the same list with the shape of each one drawn. | Flow | Roles | What it does | @@ -1217,7 +1219,11 @@ shape of each one drawn. | `humanize1:gen-plan` | `planner`, `analyst` | Turns that draft into a plan both sides have converged on. | | `humanize1:rlcr` | `builder`, `reviewer`, `human` | Builds the plan under review until nothing is left to say. Run it in a git repository. | | `parallel_flame_chase` | `coordinator`, `lane_1_actor_a` … `lane_3_actor_b`, `human` | A coordinator plans three isolated lanes; six actors alternate two to a lane and coordinate by durable report. | -| `parallel_flame_chase_git_pr` | `orchestrateor`, `lane_1_actor_a` … `lane_3_actor_b`, `human` | The same three lanes, each in a clone of its own, landing work through pull requests that are merged only when an evaluator's receipt says they improve `main`. | +| `parallel_flame_chase_git_pr` | `orchestrator`, `lane_1_actor_a` … `lane_3_actor_b`, `human` | The same three lanes, each in a clone of its own, landing work through pull requests that are merged only when an evaluator's receipt says they improve `main`. | +| `ralph_loop_agent_cleanup` | `agent`, `cleaner`, `human` | `ralph_loop`, with a cleaner that distills the workspace every few turns. Every agent role is declared with `SteeringAgentMixin`. | +| `flame_chase_agent_cleanup` | `first_chaser`, `second_chaser`, `cleaner`, `human` | `flame_chase`, with the same cleaner. | +| `recursive_lean_prover` | `worker`, `reviewer` | A Lean theorem proved by recursive decomposition, each node planned and built by `humanize1`'s phases in a worktree of its own. | +| `aot` | `writer`, `critic`, `human` | Writes a flow from a description, and lands it only once it has loaded, run on fakes and been read by a critic. | A `human` role is [the person at the prompt](#the-person-at-the-prompt) and a `workspace` the directory the run was started in, both filled by the runtime. None of them declares a budget of diff --git a/docs/reference/providers.md b/docs/reference/providers.md index 6ad8eceb..04291aff 100644 --- a/docs/reference/providers.md +++ b/docs/reference/providers.md @@ -347,7 +347,7 @@ It is *falls back to* on the menu **enter** opens, which offers that backend's o and one call apiece from Python: ```python -from hmz import providers +from hmz.coganchor import providers providers.points("claude", "subscription", "key") providers.points("claude", "key", "gateway") diff --git a/docs/reference/sdk.md b/docs/reference/sdk.md index 8338ceea..16a81685 100644 --- a/docs/reference/sdk.md +++ b/docs/reference/sdk.md @@ -65,25 +65,38 @@ a workspace is what loads the tracer. | --- | --- | | `backends()` | Every coding agent CLI humanize drives, as `hmz.coganchor.backends.Profile`. | | `reports()` | Starts [reporting humanize's own failures](/user/reporting) where that has been answered yes. Returns whether anything is being reported. | -| `read(argv)` | Reads an `hmz exec` line into what it says: the flow, an agent and an environment per role, the params, the budget, the task, and whether to pick a run up. Everything [refused before anything runs](/reference/cli#what-is-refused-before-anything-runs) is refused here. | -| `run(...)` | A [`Run`](#run) of a flow over what a line said — the agents and environments by role, the params and the budget — which is what `exec` makes of one. | +| `read(argv)` | Reads an `hmz exec` line into a `Line`: the flow, the `-a`, `-e` and `-p` it gave, the budget, the task, whether to `--resume` and whether `--json`. Only the line — nothing is loaded, and a line argparse cannot read raises `SystemExit`. | +| `runner(flow, *, agents=(), envs=(), params=None, budget=None, resume=False)` | The flow loaded, with a driver for every role it was given and everything checked — `hmz.runtime.runner.Runner` — and nothing started. Everything [refused before anything runs](/reference/cli#what-is-refused-before-anything-runs) raises `Refused` here. | +| `run(flow, task, *, agents=(), envs=(), params=None, budget=None, resume=False, outworlder=None)` | A [`Run`](#run) of that runner and the task, which is what `exec` makes of a line. `agents` and `envs` are `-a` and `-e` specs by role — `{"builder": "claude/claude-opus-5:high"}` — or drivers, or what `read` read; `params` a mapping or the flow's own model; `budget` a `Budget` or a mapping, required for every flow humanize does not ship; `resume` `True` for the newest run of the flow here that can be picked up, or the epic to pick up; `outworlder` whoever is outside the run, `None` for nobody. | | `exec(argv)` | The whole of `hmz exec`: reads the line, loads the flow, runs it to its return. | +`Refused` — `hmz.sdk.Refused`, a `ValueError` — is a run refused before anything of it ran: a +line or a setup to correct, its message saying what and its cause the exception it was refused +for. `hmz.sdk.fakes` is the in-memory kit a flow is [tested](/weaver/testing-flows) on, +`hmz.runtime.flowing.fakes`, handed through whole. + ## `Run` -One run of one flow. Making one starts nothing — whoever made it says which of the two they are -holding. +One run of one flow: `Run(runner, task, *, outworlder=None)`. Making one starts nothing — +whoever made it says which of the two they are holding. | | | | --- | --- | -| `agents` | Every agent it drives, by role. | +| `flow`, `ref`, `task` | The flow as it was named, its canonical ref, and what it was asked to do. | +| `declaration`, `budget` | What the flow declares, and what the run may spend. | +| `usage` | What every session of the run has spent so far, as a `Usage` — what its budget is held to. | +| `agents` | The coganchor agent behind each session of the run still open, oldest first, each named for its role. | +| `epic` | The [epic](/reference/tracing#epics) the run is written into, once it has started. | | `running` | Whether the flow is still going. `False` before it is started. | -| `raised` | Whatever the flow raised, for a run started on a thread and now over. | +| `raised`, `result` | Whatever the flow raised, or returned, for a run started on a thread and now over. | +| `watch(listener)` | Has everything every session says reach `listener` — the agent, the conversation and the event — from whichever thread a CLI is read on. | +| `opened(callback)` | Has each session told to `callback` as it opens: the role, the coganchor agent and its conversation. | +| `unreadable()` | Which cap of the budget nothing the run drives can read — a cost cap over a model nobody prices — in words, or `""`. | | `run()` | Runs the flow here, until it returns. | | `start()` | Runs it on a thread of its own, and returns at once. | | `wait(timeout=None)` | Waits for it to end. Returns whether it has. | -| `stop()` | Stops the flow: the turn running now is cut off, and every flow call of the run raises `FlowCancelled` at its next step rather than handing on. | -| `close()` | Closes every conversation still open, which is the backend's process going. The last thing there is to do about a run. | +| `stop()` | Stops the flow: the turn under way is interrupted and the flow unwinds — every call raises where it stands, every session it opened is closed and every temporary directory it made is taken away, in its own time. From any thread. | +| `close()` | Stops the flow and ends every conversation still open, without waiting for it: what the flow gets back is a turn that failed. The last thing there is to do about a run. | ```python from hmz.sdk import Hmz @@ -105,7 +118,8 @@ then `stop()` or `wait()` — made by `run(...)` from what `read(argv)` read off | `all()` | Every flow there is to run, by the name `-f` takes. | | `find(named)` | The file one flow is written in — or `named` itself where nothing answers to it, so whether a flow is there is whether what comes back is a file. | | `about(named)` | The line a flow says about itself. | -| the roles | What it declares: an agent role and an environment role apiece — which of them the runtime fills, which may be left out, and what each must be able to do — its [params](/reference/flows#settings-of-the-flow-s-own), and whether it [can be picked up](/user/resuming). What `/flow` asks its questions from. | +| `declared(named)` | What it declares: an agent role and an environment role apiece — which of them the runtime fills, which may be left out, and what each must be able to do — its [params](/reference/flows#settings-of-the-flow-s-own), and whether it [can be picked up](/user/resuming). What `/flow` asks its questions from. | +| `resumes(named)` | Whether it can be picked up. | | `fork(named, into=None)` | Copies it into this project's own flows, whole. | | `running()` | Every flow call of the run going now, each with its depth and the call it is under. | | `verses` | [Where flows come from](#flowverses). | @@ -196,6 +210,8 @@ happened. | `sessions(epic)` | Every session it opened. | | `opened(epic)` | What each agent opened, by the name the run knew that agent as. | | `resumed(flow)` | The newest run of one flow here that can be picked up — what `--resume` carries on. | +| `picks_up(epic)` | Whether a run can be picked up from one epic: whether its journal holds a flow call. | +| `state(epic, flow="")` | What a resumable flow kept in one run, as its journal left it — the flow the run was of, or another by its canonical ref. | | `traced(epic, *, output=None, start=None, end=None)` | Gathers one run into a [trace](/reference/tracing) of that run — its own sessions, by the ids it wrote down, beside the programs it profiled — and answers with where it went and what is in it. It goes beside the run unless an output is named. | | `trace(*, sessions=None, agents=None, output=None, start=None, end=None, profile=None)` | The same collector, asked for whatever sessions you name — which is how a session no run ever drove is read back. | | `bundled(epic, *, output=None, transcript=None)` | Packages one whole run up as one archive to send somewhere — its own records, every session log the backends wrote for it with the links followed, and a manifest — and answers with where it went and what went in. Credentials are struck out of every byte. See [Exporting a run](/user/export). | diff --git a/docs/reference/tracing.md b/docs/reference/tracing.md index e5a08d2f..cb244bc2 100644 --- a/docs/reference/tracing.md +++ b/docs/reference/tracing.md @@ -95,7 +95,7 @@ Every run of a flow is one **epic**, written as it happens, and an epic is a dir ~/.humanize/epics//-/ epic.jsonl what happened, a line at a time epic._.jsonl the same, for one flow the run called - the journal what a flow that can be picked up did, for --resume + resume.jsonl what a flow that can be picked up did, for --resume profile.jsonl the programs it ran, for a run that was profiled sessions//… a link per file the backend logged that session to traces/export.trace.json the trace exporting the run gathers, replaced each time @@ -111,10 +111,11 @@ as it goes — a run that died is a run whose epic still says what it got to. | `event` | Written | Carries | | --- | --- | --- | -| `began` | when the flow starts | `flow`, `task`, `workspace`, whether the flow is `resumable`, the run it was `picked_up` from where there was one, and one entry per agent with the role it filled, its `backend`, `model`, `effort`, `permission` and `provider` | +| `began` | when the flow starts | `flow` as it was named and its canonical `ref`, `task`, `workspace`, whether the flow is `resumable`, the run it was `picked_up` from where there was one, one entry per agent role with its `agent`, `backend`, `model`, `effort` and `provider`, the `envs` as `-e` spells each, the `params` and the `budget` | | `opened` | each time an agent opens a session | `agent`, `backend`, `provider`, `session`, the `name` the run gives it and `where` its links are | -| `called` | when the flow calls another flow | `flow`, `task`, and the `epic` — the record that call was written to | +| `called` | when the flow calls another flow | `flow`, by its canonical ref, the `task` it was called with, and the `epic` — the record that call was written to | | `returned` | when that call returns, however it ended | `flow` and the same `epic` | +| `usage` | as the run stops | what every session of it spent: `cost`, `output_tokens` and `seconds` | | `ended` | when the flow stops | `how`: `done`, `failed`, or `stopped` | `sessions//` is a link per file that session was logged to, named for whose session it @@ -140,9 +141,10 @@ epic is never reopened: running the flow again is another run, with sessions of another epic. That is what the journal, `resumable` and `picked_up` are for. A flow marked -`@flow(..., resumable=True)` keeps a journal while it runs — one JSON line per flow call and how -it ended, per write to a flow's `ctx.state`, per session opened, per temporary copy and scratch -directory kept — and `hmz exec --resume`, `/resume` or *resume this run* on `/epics` picks the +`@flow(..., resumable=True)` keeps a journal while it runs, `resume.jsonl` — one JSON line per +flow call and how it ended, per write to a flow's `ctx.state` (`{"t":"set",…}`, which +`hmz.runtime.epic.state` reads back), per session once its CLI has named it, per temporary copy +and scratch directory kept — and `hmz exec --resume`, `/resume` or *resume this run* on `/epics` picks the run up from it: the flow at the top carries on with what it kept, and each flow it calls picks up where it is called again with the same task, agents, environments and params. A run picked up is written down as a run of its own whose `began` line says which run it was `picked_up` diff --git a/docs/reference/tui.md b/docs/reference/tui.md index 685e230d..a7ff14ee 100644 --- a/docs/reference/tui.md +++ b/docs/reference/tui.md @@ -34,22 +34,21 @@ which is the one it opens on. Which, and how to move between them, is [below](#reading-one-agent). **Above the editor**, one line per agent the flow drives: the name the flow calls it, then what -it runs as `cli/model:effort`, then the machine its turns land on where that is not this one, -[what it may do](/user/permissions) where a rung was said about it at all, the -[account](#which-cli-and-which-account) it runs as where that is not this machine's own, and -finally what it is holding — `●` or `○` for whether it is working, how many conversations it -has open, `reading` on the agent whose transcript is on the screen, and `unread` on one that -has said something since you last looked at it. Under them, what the run has cost so far — one -figure per kind of token rather than one over the lot of them, then the money and the rate. -A `+` on a kind means some agent of the run drives a CLI that does not report it at all, so -that figure is a floor rather than the total; with one agent running nothing is marked. The rate is -**output tokens** a second, over a recent window only, so a flow that has stopped reads as -stopped — and the whole readout is worked out again every five seconds and whenever an agent -does anything, rather than only when a count lands. The money is per model, since two agents at -one model are one bill; it comes from [OpenLLMPrices](https://openllmprices.com/), fetched once -as the interface opens and kept under `~/.humanize/prices.json`; a model nobody lists shows its -tokens with nothing beside them rather than `$0.00`, and a run mixing a priced model with an -unpriced one marks its total `$1.34+`. See [Cost and rate](/user/tally). +it runs as `cli/model:effort`, the [account](#which-cli-and-which-account) it runs as where that +is not this machine's own, and finally what it is holding — `●` or `○` for whether it is +working, how many conversations it has open, `reading` on the agent whose transcript is on the +screen, and `unread` on one that has said something since you last looked at it. Under them, +what the run has cost so far — one figure per kind of token rather than one over the lot of +them, then the money and the rate. A `+` on a kind means some agent of the run drives a CLI that +does not report it at all, so that figure is a floor rather than the total; with one agent +running nothing is marked. The rate is **output tokens** a second, over a recent window only, so +a flow that has stopped reads as stopped — and the whole readout is worked out again every five +seconds and whenever an agent does anything, rather than only when a count lands. The money is +per model, since two agents at one model are one bill; it comes from +[OpenLLMPrices](https://openllmprices.com/), fetched once as the interface opens and kept under +`~/.humanize/prices.json`; a model nobody lists shows its tokens with nothing beside them rather +than `$0.00`, and a run mixing a priced model with an unpriced one marks its total `$1.34+`. See +[Cost and rate](/user/tally). **The status line, left:** what is running, if anything is — whose turn it is and how long it has been going. Between two turns it names the flow and how long the run has been going, since @@ -168,10 +167,10 @@ list appears under the editor with a line about each. | `/flow` | `[flow]` | The menu that is [which flow runs](#choosing-a-flow) and, inside the flow you open, [what fills each of its roles](#what-each-agent-is), its params and its budget. With a name or a path, opens already inside that one — and is refused outright while a flow is running, since that name would be choosing one. Without a name it opens on the flows, or inside the agents of the flow that is going. `save` — the row set below the choices, and **shift+enter** or **ctrl+j** from anywhere on the menu — saves the complete setup; esc is one step back, and then the way to save or discard on the way out. | | `/flowverses` | | [Where flows come from](/weaver/flowverses): what places there are, what one of them holds, and one added, fetched again or taken away. The same menu **v** opens on the flows; a command as well, because there are no flows to press it on while one is running. Not which flow to run — that is `/flow`, where the arrows step between the same places. | | `/epics` | | The runs of this directory, newest first: what each was and how it went. **Enter** goes into one, which says where it is written down and offers [exporting it](/user/export) — trace and all — and carrying it on where its flow says it can be picked up. | -| `/resume` | | [Picks up](#carrying-the-last-one-on-outright) the newest run of the flow in force here that can be picked up: that run's own flow, on its own agents and environments, with its params and what it was asked to do, from where its journal says it got to. The same thing `/epics` offers of the run you go into, without the list. Where there is nothing to pick up it says which reason that is. | +| `/resume` | | [Picks up](#carrying-the-last-one-on-outright) the last run here of a flow that can be picked up: that run's own flow, on its own agents and environments, with its params, its budget and what it was asked to do, from where its journal says it got to. The same thing `/epics` offers of the run you go into, without the list. Where there is nothing to pick up it says which reason that is. | | `/providers` | | [The accounts](#the-accounts-themselves) an agent may be run as: what there is, and what can happen to one — made, and, on enter, corrected, signed in again, pointed at what it falls back to, or taken away. How often a failed turn is taken again is not here: that is said of a [place](#where-a-turn-goes-when-it-cannot-be-taken) rather than of an account. | | `/settings` | | [What humanize remembers](#what-humanize-remembers): two pages, one for what is true of this machine and one for what is remembered about this directory. | -| `/monitor` | | [The run, drawn](#watching-the-run): a box per agent that has worked, marked as it works and saying how long it has been at it, with the handovers between them as the arrows joining them, whatever each started of its own hanging under it, and [the board](/user/board) below. Enter reads an agent or changes a line. **esc** opens it. | +| `/monitor` | | [The run, drawn](#watching-the-run): a box per agent that has worked, marked as it works and saying how long it has been at it, with the handovers between them as the arrows joining them, whatever each started of its own hanging under it, and [the board](/user/board) below where the run has one. Enter reads an agent or changes a line. **esc** opens it. | | `/btw` | `` | Asks a side question about the running flow from a read-only snapshot of its progress. It runs in a separate session and never steers the flow. | | `/details` | `[on\|off]` | Shows or hides everything a turn did on the way to its answer: tool calls, thinking, and whatever a backend printed on its way past. One question — how much of the working to show — so one switch. **Off** to begin with. | | `/afk` | `[on\|off]` | Whether you are there to be asked. While it is on, the run's [outworlder](/reference/flows#the-person-at-the-prompt) is away. See [below](#questions-and-being-away). | @@ -197,7 +196,7 @@ morning. $ralph_loop fix the failing test ``` -What happens next depends on whether this directory has run that flow before: +What happens next depends on whether that flow is set up here: | | | | --- | --- | @@ -206,15 +205,15 @@ What happens next depends on whether this directory has run that flow before: | **No such flow** | A line to correct, the way `/nosuchcommand` is. The interface stays up. | **Set up here** means a remembered agent for every agent role the flow declares *now* and a -remembered environment for every environment role — under the name the flow declares each -role by — params the flow still accepts, and a budget. Roles the runtime fills — an -`Outworlder`, a `LocalEnv` — need nothing remembered, and a `NotRequired` role left empty is -set up as empty. A flow that has grown, lost or renamed a role since is one you are asked about -again, rather than one whose new reviewer quietly inherits the builder's model. So are params -that no longer read back through the `FlowParams` the flow declares now: nothing can guess what -an answer that no longer fits was meant to say. A flow you never set any params for is not one -of those — it takes its own defaults, exactly as `hmz exec` does with no `-p`. A budget is -never assumed: a flow with none remembered is asked for one, `chat` excepted. +remembered environment for every environment role — under the name the flow declares each role +by — params the flow still accepts, and a budget. Roles the runtime fills — an `Outworlder`, a +`LocalEnv` — need nothing remembered, and an environment role declared `NotRequired` and left +empty is set up as empty. A flow that has grown, lost or renamed a role since is one you are +asked about again, rather than one whose new reviewer quietly inherits the builder's model. So +are params that no longer read back through the `FlowParams` the flow declares now: nothing can +guess what an answer that no longer fits was meant to say. A flow you never set any params for +is not one of those — it takes its own defaults, exactly as `hmz exec` does with no `-p`. A +budget is never assumed: a flow with none remembered is asked for one, `chat` excepted. `$` takes a **name**, not a path: a path holds the slashes, dots and spaces prose does, and one taken here would swallow the line after it. `/flow ./flows/mine` is where a flow of your own by @@ -320,7 +319,7 @@ the transcript they all appear on. open. Where you are reading all of them at once there is no one agent you can have meant, so it goes to whichever has a turn open, which is the one the screen is showing anyway. -What is kept is bounded, a flow being a thing that runs for days: the last eight conversations, +What is kept is bounded, a flow being a thing that runs for days: the last sixteen transcripts, and the last two thousand lines of each. Older lines and older conversations are gone from the screen, not from the [trace](/reference/tracing) — that is what a trace is for. @@ -351,10 +350,6 @@ neighbours as the arrows joining them: │ codex/gpt-5.6-sol:high · 5 turns unread │ └────────────────────────────────────────────────────────┘ - Board · what you and the flow both write on - ◈ todo write the parser - ◈ doing two of five · flow's - Flow: chat Also: builder → reporter · ×2 Tokens: claude-opus-5 48.2k $1.34 91 out/s @@ -407,8 +402,9 @@ fleet too long to draw is cut, with a line saying how many were left off. which nobody waits at. `a` puts one up — a name, then what it says — enter changes the one under the cursor, and `d` twice takes it off — the one taking-away still on a key, because enter on a line opens the words of that line rather than a menu with a row to spare, and it lands the -moment it is pressed. The flow API gives a flow no way to read or write it, so a run of a flow -written against it has nothing there of the flow's. See [The mission board](/user/board). +moment it is pressed. The flow API gives a flow no way to read or write it, and a run of a flow +written against it has no board at all: none is drawn, and `a` says there is none. See +[The mission board](/user/board). **Enter or a click on a box reads that agent** — whether or not it is working. tab is held to the ones thinking, so this is the one place an agent that has stopped is reached. The box under @@ -434,7 +430,7 @@ what the run is running as, and the two are read from the bottom up — the last the running total end on the same row: ``` - assistant · claude-opus-5:high + assistant · claude/claude-opus-5:high ❯ and fix the tests too input 11.2k · output 1.1k · cache_read 0 ❯ then push $0.31 · 84 out/s ──────────────────────────────────────────────────────────────────────── @@ -511,10 +507,11 @@ turn — the status line says `enter answer` while that is so. questions are answered at once with nothing — `""` for text, the answer a shape's defaults make where every field has one, and `OutworlderAway` raised in the flow where one has none — and an agent's question put to you is told nobody answered and carries on, rather than waiting on a -reply that is not coming. Asking starts **allowed**: a flow that really needs a person gets one -unless it has been said that none is there. While it is on, the status line says `afk` in front -of everything else on it: the one sign that a question went unanswered must not be a flow that -finished early. +reply that is not coming. A question already up when it goes on is answered by nobody at all, +which the flow hears as `OutworlderAway`. Asking starts **allowed**: a flow that really needs a +person gets one unless it has been said that none is there. While it is on, the status line says +`afk` in front of everything else on it: the one sign that a question went unanswered must not +be a flow that finished early. A question still up when the flow ends or is stopped ends with it, so stopping a flow is never blocked on one. @@ -765,10 +762,10 @@ steering, a hook only some CLIs reach — are the flow's, [declared on the role' type](/reference/flows#asking-for-an-agent-that-can-do-something). None of them is a row here, because a row offering to set one would be a second answer to a question already settled. -**Roles the runtime fills are listed and not asked.** An `Outworlder` role is you — see +**Roles the runtime fills are not listed at all.** An `Outworlder` role is you — see [Questions, and being away](#questions-and-being-away) — and a `LocalEnv` role is the directory -the interface was started in; both are shown as filled. A role the flow declares `NotRequired` -may be left empty. +the interface was started in; neither is a row. An environment role the flow declares +`NotRequired` may be left empty. The rows are in the order of what depends on what. The CLI settles which accounts there are and which models that CLI will name; the account settles which of them it may name. **Changing the @@ -817,8 +814,8 @@ it: it asks how to sign in and what that way needs — the same walk account chosen. A CLI with no accounts yet says `claude has no accounts here yet` under the list, with the `add` row under that. -An agent given an account that has since been taken away is a red line when the flow is started, -before any turn has run — never a traceback half an hour in. +An agent given an account that has since been taken away never quietly runs as yours: the first +turn it takes fails, naming the account that is not there. ## What each agent runs @@ -859,10 +856,11 @@ the agents, answered with where that directory is — what an `-e` says: | `ssh@gpu-box/home/me/repo` | a directory on a host you can reach with ssh | | `ssh@gpu-box/~/repo` | the same, under the home directory of whoever ssh logs in as | -The sheet lists the hosts with an entry in your `~/.ssh/config`, and anything else is a -destination you type after **s** — `host`, `user@host`, `host:port`. A role typed as a -`LocalEnv` is the directory the interface was started in and is not asked; most flows work in -nothing else, and have no environment rows at all. +Enter on the row opens one field, typed as `-e` spells it after `=` — the host `host`, +`user@host`, `host:port` or an alias of your ssh config — and read as `-e` reads it: one that +does not read is said in red under the roles rather than taken, and an empty one leaves the role +unanswered. A role typed as a `LocalEnv` is the directory the interface was started in and is +not asked; most flows work in nothing else, and have no environment rows at all. A host that cannot be reached, a directory that is not there, and a machine smaller than the role declares — fewer CPUs or GPUs, less memory — are red lines when the flow is started, @@ -892,13 +890,13 @@ and this is what reads it back: do and how many sessions it opened, the newer one marked "can be picked up"](/demo/epics.png) A row is when the run began and the flow that ran; beside it, what that flow was asked to do, -how many sessions it opened, and `can be picked up` for a run whose flow said it was -resumable. How it went is there only where it went some way other than finishing — stopped, -failed, or left unfinished by a machine that went away under it — since a list of runs is -mostly runs that finished, and a column saying so of nearly all of them is a column taking the -room the others need. Newest first, because what somebody who opens this came to look at is -the run that has just happened. **s** searches the flow, what it was asked to do, and the name -the run is written under. +how many sessions it opened, and `can be picked up` for a run whose flow says it can be and that +left a journal to pick up from. How it went is there only where it went some way other than +finishing — stopped, failed, or left unfinished by a machine that went away under it — since a +list of runs is mostly runs that finished, and a column saying so of nearly all of them is a +column taking the room the others need. Newest first, because what somebody who opens this came +to look at is the run that has just happened. **s** searches the flow, what it was asked to do, +and the name the run is written under. The list is read rather than chosen from, so **enter** goes *into* the run under the cursor. What opens says where that run is written down — sessions and all, which is what anybody @@ -909,7 +907,7 @@ export it](/demo/epic-does.png) | Row | What it does | | --- | --- | -| **resume this run** | Picks that run up — its own flow, on its own agents and environments, with its params — from where its journal says it got to, which a flow that says it [can be picked up](/reference/flows#a-flow-that-can-be-picked-up) keeps. It is [`/resume`](#carrying-the-last-one-on-outright) with the run already named, and a run that cannot be picked up is turned down here in the same words. | +| **resume this run** | Picks that run up — its own flow, on its own agents and environments, with its params and its budget — from where its journal says it got to, which a flow that says it [can be picked up](/reference/flows#a-flow-that-can-be-picked-up) keeps. It is [`/resume`](#carrying-the-last-one-on-outright) with the run already named, and a run that cannot be picked up is turned down here in the same words. | | **export it** | Packages **that run** up as one archive to send to somebody else — its own records, every session log the backends wrote for it as their contents rather than as the links the run keeps, a manifest, and a [trace](/user/tracing) of the run gathered on the way in. There is no transcript in this one: what is on your screen is not that run. Where it landed and how big it came out are said under the list, and again in the transcript. See [Exporting a run](/user/export). | **Exporting gathers the trace.** They were two rows, and one of them wrote a file into the run @@ -921,17 +919,18 @@ run in fifty times has fifty traces and none of them holds another's work. The t the run's own `traces/` as well, rather than in whatever directory you are standing in. **Carrying on is offered where the flow says so now**, rather than where the run said so then. -The mark on the row is what that run wrote down as it ran; going into the run asks the flow -itself, since a flow is a file that may have been rewritten since — and one that will not load -at all is one there is nothing to carry on from. Where it is not offered the row is not there -and the reason is said under the one that is. Exporting is offered for every run, whatever its -flow says: a run that cannot be continued is still a run to read. +The mark on the row and the row inside the run both ask the flow itself, since a flow is a file +that may have been rewritten since — and one that will not load at all is one there is nothing +to carry on from. Where it is not offered the row is not there and the reason is said under the +one that is. Exporting is offered for every run, whatever its flow says: a run that cannot be +continued is still a run to read. What is carried on is the run rather than what the interface happens to be set up on — the -flow, its agents, its environments, its params and what they were asked to do all come off the -record of that run, an agent swapped under it being a different run wearing its name. The flow -at the top picks up what it kept, and each flow it calls picks up where it is called again with -the same task, agents, environments and params; the budget is a fresh one. +flow, its agents, its environments, its params, its budget and what they were asked to do all +come off the record of that run, an agent swapped under it being a different run wearing its +name. The flow at the top picks up what it kept, and each flow it calls picks up where it is +called again with the same task, agents, environments and params; what the budget has spent is +counted again from nothing. **Reading is not refused while a flow is running. Carrying one on is.** What has already happened does not change under you, so the list is worth having open mid-run — but a run @@ -949,22 +948,28 @@ runs, and a trace of none of them has nothing here to hang on. ### Carrying the last one on outright -`/resume` is **resume this run** without the list: it picks up **the newest run of the flow in -force here that can be picked up** — the one somebody who left a loop running overnight came -back for, and what `hmz exec --resume` picks up from a command line. The flow, its agents, its -environments, its params and what they were asked to do come off that run exactly as they do -from inside it, and the line it starts on says which run is being picked up — a person who has -been away is owed which day's work this is. - -Where there is nothing to pick up, the reason is said instead. These are the same reasons in the -same words for a run you walked into on `/epics` and took **resume this run** on — one question -has one answer, whichever way you came to it: +`/resume` is **resume this run** without the list: it picks up **the last run here of a flow +that can be picked up**, whichever flow that was — the one somebody who left a loop running +overnight came back for. Runs of a flow that neither said nor says it can be picked up — a +conversation had since — are passed over, and nothing further back is: where the last run of a +flow that can be picked up cannot be, it says why rather than handing over the one before it. +`hmz exec --resume` is the nearest thing on a command line: the newest run of the flow its `-f` +names that left a journal. The flow, its agents, its environments, its params, its budget and +what they were asked to do come off that run exactly as they do from inside it, and the line it +starts on says which run is being picked up — a person who has been away is owed which day's +work this is. + +Where there is nothing to pick up, the reason is said instead. From ` cannot be read back` +down, these are the same reasons in the same words for a run you walked into on `/epics` and +took **resume this run** on — one question has one answer, whichever way you came to it: | | | | --- | --- | -| `no run of here can be picked up` | Nothing has run that flow in this directory, or nothing that ran it kept a journal. | +| `no flow has been run here` | Nothing has run in this directory at all. | +| `no run here was of a flow that can be picked up` | Every run here was of a flow that neither said nor says it can be picked up. | | ` cannot be read back` | Its record is not one — a run that died mid-line left a line rather than an epic. | | ` does not say it can be picked up` | Asked of the flow as it is today, not of what the run recorded — and a flow that will not load at all reads as one that says no. | +| ` left nothing behind` | It was killed before its journal held anything. Say what to do and the flow starts from the top. | | `no picking a run up while a flow is running` | A run picked up is a flow started, and there is one going. [ctrl+c twice or `/stop`](/user/stopping) stops it first. | | `no picking a run up while the flow is still stopping` | ctrl+c twice has been pressed and the flow has not gone yet. It unwinds in its own time, so a run picked up from a journal still being written is a round done twice. A flow that will not unwind at all is what the [third press](/user/stopping) is for. | @@ -1113,7 +1118,8 @@ environment that is answering for this run says so under the list rather than be the setting, since a menu cannot change it. **This directory** is what is remembered here: the directory itself, the flow it opens on and -how many roles that flow was set up with, and a row that forgets the lot — leaving every other +how many agents that flow was set up with, whether a run here is +[profiled](/user/tracing#profiling-a-run), and a row that forgets the lot — leaving every other directory, and every setting, as it was. The arrows and space step the row under the cursor, and nothing lands until the `save` row is diff --git a/docs/tapes/run.tape b/docs/tapes/run.tape index 30dcd308..89eb1eea 100644 --- a/docs/tapes/run.tape +++ b/docs/tapes/run.tape @@ -38,6 +38,7 @@ Sleep 3s Screenshot "/out/run-linked.png" Sleep 600ms -# What a flow that says it can be picked up left for the next run of it. -Type 'cat "$run"state.json' Enter +# What a flow that says it can be picked up left for the next run of it: the engine's +# journal, whose `set` lines are the state `--resume` hands back. +Type 'cat "$run"resume.jsonl' Enter Sleep 2800ms diff --git a/docs/tapes/stage.py b/docs/tapes/stage.py index f2e2f067..d971c473 100644 --- a/docs/tapes/stage.py +++ b/docs/tapes/stage.py @@ -11,6 +11,7 @@ from __future__ import annotations +import asyncio import datetime import json import pathlib @@ -355,8 +356,8 @@ def _runs() -> None: Invented, like everything else here -- the moments included, so that a rendered GIF says the same date tomorrow. What is not invented is the shape: this is `hmz.runtime.epic` writing - its own record, linking each session to the transcript above and keeping what a flow that - can be picked up left behind. + its own record, linking each session to the transcript above, and the engine's journal + keeping what a flow that can be picked up left behind. """ from hmz.coganchor.agents import AgentConfig from hmz.runtime import epic as written_as @@ -371,23 +372,75 @@ def _runs() -> None: written_as.uuid = _Invented() config = AgentConfig(model="claude-opus-4-8", effort="high") + budget = {"duration": None, "cost": 5.0, "output_tokens": None, "graceful": True} written_as._now = _ticks(0) # noqa: SLF001 -- each run happened when it happened first = _Drove(id="builder", backend="claude", config=config) - with written_as.Epic("twice", [first], "work through TASK.md", WORK) as one: + with written_as.Epic( + "twice", + "work through TASK.md", + WORK, + ref="twice:twice", + agents=[written_as.Drove("builder", "claude", "claude-opus-4-8", "high")], + budget=budget, + ) as one: first.epic = one one.opened(first, SESSION) written_as._now = _ticks(LATER) # noqa: SLF001 second = _Drove(id="fixer", backend="claude", config=config) with written_as.Epic( - "nightly", [second], "keep the tests green", WORK, resumable=True + "nightly", + "keep the tests green", + WORK, + ref="nightly:nightly", + agents=[written_as.Drove("fixer", "claude", "claude-opus-4-8", "high")], + budget=budget, + resumable=True, ) as two: second.epic = two two.opened(second, NIGHTLY) - held = two.state("nightly") - held["rounds"] = 3 - held["fixed"] = ["add() subtracted", "divide() raised nothing"] + asyncio.run(_kept(two.resume)) _profile(two.path) + two.stopped() # its budget ran out, which is how a loop like this one ends + + +async def _kept(at: pathlib.Path) -> None: + """Writes what the resumable run kept into its journal, as the engine writes one. + + Args: + at: The journal, inside the epic. + """ + from hmz.runtime.flowing.journaling import Journal, digest + + journal, _ = Journal.opened(at, asyncio.get_running_loop(), resume=False) + journal.call( + 1, + 0, + digest( + "nightly:nightly", + "keep the tests green", + ["fixer=claude@/claude-opus-4-8:high"], + b"{}", + ), + 0, + b'"nightly:nightly"', + ) + journal.set(1, "rounds", b"3") + journal.set( + 1, "fixed", json.dumps(["add() subtracted", "divide() raised nothing"]).encode() + ) + journal.note( + { + "t": "session", + "id": 1, + "role": "fixer", + "harness": "claude", + "model": "claude-opus-4-8", + "session": NIGHTLY, + } + ) + journal.end(1, ok=False) + journal.close() #: What the invented run is to have spent its minutes on: an agent's turn is mostly other @@ -466,7 +519,11 @@ def _settings() -> None: Settings(WORK).profiles(on=True) # And what this project was last set up to run, so that what humanize remembers about a # directory is a directory it has been used in. - Settings(WORK).remember("twice", ("",), [Runs("claude/claude-opus-4-8:high")]) + Settings(WORK).remember( + "twice", + {"builder": Runs("claude/claude-opus-4-8:high")}, + budget={"cost": 5.0}, + ) if __name__ == "__main__": diff --git a/docs/user/afk.md b/docs/user/afk.md index 56c3b0ed..66aaaf87 100644 --- a/docs/user/afk.md +++ b/docs/user/afk.md @@ -50,7 +50,8 @@ the answer, rather than a word in the turn. The agent's offer appears with it, b not limited to those options — every backend that offers them also takes something else. A question still up when the flow ends or is stopped ends with it, so stopping a flow is never -blocked on a question. +blocked on a question. One still up when you turn `/afk` on is answered by nobody at all, which +the flow hears as `OutworlderAway`. ## On a command line diff --git a/docs/user/concepts.md b/docs/user/concepts.md index 06dcca4c..9cf60b9c 100644 --- a/docs/user/concepts.md +++ b/docs/user/concepts.md @@ -147,8 +147,8 @@ and three to run. Each asks only for the roles it drives. At the prompt a flow is named by that same name, and a `$` in front of it [starts one outright](/reference/tui#starting-a-flow-outright): `$ralph_loop fix the failing -test` is that flow, run on that line — the menu only opens if this directory has never set -that flow up. +test` is that flow, run on that line — the menu only opens if that flow is not set up in this +directory. See [Flows](/reference/flows). diff --git a/docs/user/conversations.md b/docs/user/conversations.md index eef80dd1..e65452e5 100644 --- a/docs/user/conversations.md +++ b/docs/user/conversations.md @@ -23,7 +23,7 @@ With ten agents going, these step between the ones thinking right now, not the o stopped. An agent between its turns stays readable once you are on it — what you are reading stays put until you press one of these keys — but it is not stepped onto. -**Every agent there is can still be read**, from the diagram +**Every agent that has worked can still be read**, from the diagram [`/monitor`](/reference/tui#watching-the-run) draws. **esc** opens it, and enter or a click on a box reads that agent whether or not it is working. That is where the one that has stopped, or has not started, is picked out by name rather than stepped past. diff --git a/docs/user/efforts.md b/docs/user/efforts.md index b12496fd..de778f67 100644 --- a/docs/user/efforts.md +++ b/docs/user/efforts.md @@ -25,8 +25,8 @@ the `swarm` row turns swarm mode on for a model that has one. ## Set the effort `backend/model:effort` is how an agent carries one on a command line — the effort last, after -the last colon, however many slashes the model's own name has in it. A flow's Python config -takes the same word: +the last colon, however many slashes the model's own name has in it. An agent configured in +Python takes the same word: ::: code-group @@ -140,5 +140,5 @@ hold the agent to a target. A flow cannot: the flow API hands a flow an agent's ## See also - [Cost and rate](/user/tally) -- [Permissions](/user/permissions) — the other thing set on the model sheet +- [Permissions](/user/permissions) — what an agent may do, which its flow says - [Agents › Efforts](/reference/agents#efforts) diff --git a/docs/user/export.md b/docs/user/export.md index ba8b9301..534bfcb2 100644 --- a/docs/user/export.md +++ b/docs/user/export.md @@ -16,7 +16,7 @@ carries what is behind it. [`/epics`](/user/tracing#what-a-run-writes-down) lists every run of this directory, newest first. Enter on one opens what there is to do with it, and **export it** is there beside -collecting its trace — a run is exported where the runs are, rather than by a command about +resuming it — a run is exported where the runs are, rather than by a command about whichever one your screen happens to be showing. Then look at what it names: ```sh @@ -31,7 +31,7 @@ Everything the run wrote down about itself, and everything its sessions were log | --- | --- | | `epic.jsonl` | what happened, a line at a time: which flow, on what, by which agents, which session each of them opened, and how it ended | | `epic._.jsonl` | the same again for every flow this run [called](/weaver/calling-flows) — one run, however many flows it took | -| the journal | what a [resumable](/user/resuming) flow did, for picking it up | +| `resume.jsonl` | what a [resumable](/user/resuming) flow did, for picking it up | | `profile.jsonl` | the programs it ran, for a [profiled](/user/tracing#profiling-a-run) run | | `traces/…` | every [trace](/user/tracing) gathered of it | | `sessions//…` | the backends' own logs, **as their contents** rather than as the links the run keeps — one directory per session, named for the agent, the CLI, the account and the id | @@ -40,9 +40,9 @@ Everything the run wrote down about itself, and everything its sessions were log The manifest is what makes the rest readable by somebody who was not there — humanize's own version, Python's and the machine's; the flow, the task and how it went; one line per agent -saying the CLI, the model, the effort, what it was allowed and the account **by name**; the -workspace and the commit it is on if it is a git repository; and, per backend, the version it -says it is and the SHA-256 of the executable that actually took the turns. These CLIs move +saying the CLI, the model, the effort and the account **by name**; where each environment was; +the workspace and the commit it is on if it is a git repository; and, per backend, the version +it says it is and the SHA-256 of the executable that actually took the turns. These CLIs move weekly, so a bug is a bug in a build. It also says the shape the run ran in: every flow it called, however deep, each naming the diff --git a/docs/user/installation.md b/docs/user/installation.md index dc7812ad..30a6e50b 100644 --- a/docs/user/installation.md +++ b/docs/user/installation.md @@ -229,7 +229,7 @@ Nothing is written until something needs it. | `~/.humanize/history.jsonl` | what has been typed at the prompt | | `~/.humanize/flowverses/` | the [flowverses](/weaver/flowverses) fetched here | | `~/.humanize/providers/` | the [accounts](/user/providers), `0600` in a `0700` directory | -| `.humanize/` in a project | exported transcripts, and this project's own flows | +| `.humanize/` in a project | [exported runs](/user/export), and this project's own flows | `HUMANIZE_HOME` moves the first five somewhere else. The full list is in the [CLI reference](/reference/cli#files). diff --git a/docs/user/monitor.md b/docs/user/monitor.md index 317e3205..9121f715 100644 --- a/docs/user/monitor.md +++ b/docs/user/monitor.md @@ -9,7 +9,8 @@ first glance. The diagram is the sheet, not a header on one. It takes the height your terminal has, and the few lines under it are only what a picture cannot say. -It is also where [the board](/user/board) is, where a run has one. +It is also where [the board](/user/board) is, where a run has one — which a run of a flow +written against the flow API never does. ## Try it @@ -115,9 +116,8 @@ graph and spin a clock at them while they thought. Little of this waits for `/monitor`. Three parts of the screen carry it while the run goes on. **Above the editor**, continuously: one line per agent. Each line shows the name the flow calls -it, what it runs as `cli/model:effort`, the machine, [what it may do](/user/permissions) and -the account where those are not the ordinary ones, and how many conversations it holds. `●` is -an agent with a turn open, `○` one that has stopped. +it, what it runs as `cli/model:effort`, the account where that is not this machine's own, and +how many conversations it holds. `●` is an agent with a turn open, `○` one that has stopped. **On the status line, left**: whose turn it is and how long it has been going; between turns, the flow and how long the run has been going. A flow that [called @@ -160,7 +160,7 @@ Which flows are running, innermost last: ```python from hmz.runtime.flowing import running -running() # one LiveCall(ref, name, depth, since, id, parent) apiece +running() # one LiveCall(ref, name, depth, since, id, parent, task, resumable) apiece [one.name for one in running()] # ["chat", "rlar"] ``` diff --git a/docs/user/providers.md b/docs/user/providers.md index 5a679958..3e618e3d 100644 --- a/docs/user/providers.md +++ b/docs/user/providers.md @@ -62,8 +62,8 @@ In Python the account is a field of the config: ClaudeCodeAgentConfig(model="claude-opus-5", effort="max", provider="deepseek") ``` -At the prompt it is the **account** row of the sheet an agent is set up on, which is the second -page of `/flow`. It sits under the `cli` row, because an account belongs +At the prompt it is the `provider` row of the sheet an agent is set up on, which enter on one of +a flow's roles in `/flow` opens. It sits under the `cli` row, because an account belongs to one backend: what signs in to Claude Code is not what signs in to codex. Opening it lists that CLI's own accounts with `as local` first: @@ -336,8 +336,8 @@ $ hmz exec -f ralph_loop -a agent=claude@gone/claude-opus-5:max -b cost=5 "…" … no claude provider called 'gone' ``` -In the interface, an agent given an account that has since been taken away is a red line when -the flow is started, before any turn has run. +In the interface it is the same: an agent given an account that has since been taken away +fails the first turn it takes, naming the account that is not there. ## When one goes down diff --git a/docs/user/remote-execution.md b/docs/user/remote-execution.md index 31ba0ef4..fb731901 100644 --- a/docs/user/remote-execution.md +++ b/docs/user/remote-execution.md @@ -218,8 +218,8 @@ hmz exec -f onbox -a builder=claude/claude-opus-5:high -a reviewer=codex/gpt-5.6 A role typed as a `LocalEnv` — `workspace` above — is the directory the run started in, and is never named on the line. The flow's own `await envs["box"].exec([...])`, `read` and `write` run on the box too, so the flow reads what the agent did where the agent did it. At the prompt the -same answer is a row of the flow's sheet in [`/flow`](/reference/tui#choosing-a-flow), listing the -hosts in your `~/.ssh/config`. +same answer is a row of the flow's sheet in [`/flow`](/reference/tui#where-each-agent-works), +typed as `-e` spells it. **Outside a flow**, give an agent's config an anchored machine and its turns land there: diff --git a/docs/user/reporting.md b/docs/user/reporting.md index 7596f956..d28d98f4 100644 --- a/docs/user/reporting.md +++ b/docs/user/reporting.md @@ -34,7 +34,7 @@ question, and nothing on a CI box should start uploading because nobody was ther | | | | --- | --- | | the error | its type, its message, and where in humanize it happened | -| the run | which flow, how long it had been going, and one line per agent: the CLI, the model, the effort, the account **by name**, what it may do, where its work lands, which skills the flow mounted | +| the run | which flows were going, how deep and for how long, and one line per agent role: the CLI, the model, the effort, the account **by name**, what it may do, which skills it carries | | the machine | which coding agents are installed, which accounts exist and how each was signed in, which skills each CLI would load and which flowverses are here — all by name | | the friction | what humanize did that you then undid, refused or walked away from, as counts | | the versions | humanize, Python, and the kind of machine | @@ -45,8 +45,8 @@ question, and nothing on a CI box should start uploading because nobody was ther - **Nothing an agent said.** No transcript, no session log, no tool output. - **No file, no path outside humanize itself, and no directory name.** A stack frame is named by where it sits under humanize, or under whatever humanize is installed beside — - `hmz/agents/base.py`, `textual/app.py`. A frame in anything else, such as a flow of yours, - keeps its line number and nothing else. No path, no file name, no module, no function. A home + `hmz/coganchor/agents/base.py`, `textual/app.py`. A frame in anything else, such as a flow of + yours, keeps its line number and nothing else. No path, no file name, no module, no function. A home directory is replaced wherever it appears, even inside an exception's own message. The command line a failed turn ran as is also taken out of the one line Python writes for it. For several of these backends that command line holds the prompt. @@ -95,7 +95,7 @@ a report by handing over something that knows. That thing runs only when a repor being made, so nothing is gathered on a machine that reports nothing: ```python -from hmz import telemetry +from hmz.runtime import telemetry telemetry.about("worktrees", lambda: {"held": len(worktrees)}) telemetry.snag("gave-up", after=3) # not an error, and not what anybody meant either diff --git a/docs/user/resuming.md b/docs/user/resuming.md index 40080a0d..62ad6403 100644 --- a/docs/user/resuming.md +++ b/docs/user/resuming.md @@ -64,10 +64,13 @@ picked an earlier one up. A flow that says nothing runs from the top every time. ## Where it lives -A resumable run keeps a **journal** beside its epic: one JSON line per thing a run picking it up -needs — each flow call and how it ended, each write to a flow's state, each session opened, each -temporary copy and scratch directory kept. It is appended to as the run goes, so a run that was -killed rather than stopped still says what it got to. +A resumable run keeps a **journal** inside its epic, `resume.jsonl`: one JSON line per thing a +run picking it up needs — each flow call and how it ended, each write to a flow's state (a +`{"t":"set",…}` line, and `{"t":"del",…}` for a key taken out), each session opened once its CLI +has named it, each temporary copy and scratch directory kept. It is appended to as the run goes, +so a run that was killed rather than stopped still says what it got to. There is no separate +state file: what a run kept is read back off those lines, which is what +`Hmz().epics.state(epic)` does from Python. State is kept **per call**, so a flow that calls [another one](/reference/flows#a-flow-that-calls-another-flow) is two flows, each keeping its own state @@ -98,7 +101,9 @@ reviews of the rounds it had done, and a round with a different task is a new ro Temporary copies and scratch directories a resumable run made are kept rather than removed as the flow that made them ends, so the run picking it up finds them where they were. -The budget is not picked up: a run picked up is held to the `-b` of the line that picked it up. +What the budget has spent is not picked up. On a command line a run picked up is held to the +`-b` of the line that picked it up; in the interface, to the budget the run it picks up was +given, counted again from nothing. ## Running it again @@ -111,9 +116,10 @@ changing `-p` on a `--resume` line does not start the run over; leaving `--resum ## Picking one up from the interface -**`/resume`** picks up the newest run of the flow in force here that can be picked up: that -run's own flow, on its own agents and environments, with its params and what it was asked to do. -Which one that was comes back on the line that starts it — +**`/resume`** picks up the last run here of a flow that can be picked up, whichever flow that +was: that run's own flow, on its own agents and environments, with its params, its budget and +what it was asked to do. Runs since of a flow that cannot be picked up — a conversation had in +between — are passed over. Which one that was comes back on the line that starts it — ``` carrying on from 20260910T021407.882Z-a3f19c: nightly on what that run left behind @@ -124,22 +130,24 @@ they need to know first. Where there is nothing to pick up it says which reason | | | | --- | --- | -| `no run of here can be picked up` | Nothing has run that flow in this directory, or nothing that ran it kept a journal. | +| `no flow has been run here` | Nothing has run in this directory at all. | +| `no run here was of a flow that can be picked up` | Every run here was of a flow that neither said nor says it can be picked up. | | ` cannot be read back` | Its record is not one: a run that died mid-line left a line rather than an epic. | -| ` does not say it can be picked up` | Asked of the flow as it stands today, not of what the run recorded. | +| ` does not say it can be picked up` | Asked of the flow as it stands today, not of what the run recorded. The last run here was of a flow that said so then and does not now — and a run further back is not handed over instead. | +| ` left nothing behind` | It was killed before its journal held anything. Say what to do and the flow starts from the top. | | `no picking a run up while a flow is running` | A run picked up is a flow started, and one is going. [ctrl+c twice or `/stop`](/user/stopping) stops it first. | | `no picking a run up while the flow is still stopping` | ctrl+c twice was pressed and the flow has not gone yet — it is closing out the turn it was in, and its journal is still being written. | `/resume` takes nothing after it: a line that names a run is said back rather than dropped. -To carry on a run that is **not** the newest, open the list and go into that run — which is the -next section. +To carry on any other run, open the list and go into that run — which is the next section. ## Carrying an older one on -`/epics` is every run of a flow in this directory, newest first: when it happened, which flow -it was, what it was asked to do, how many sessions it opened, and a mark on the runs whose flow -says it can be picked up. Enter goes **into** the run under the cursor — which says where that -run is written down, and offers what there is to do with it: +`/epics` is every run of a flow in this directory, newest first: when it happened, which flow it +was, what it was asked to do, how many sessions it opened, and a mark on the runs whose flow +says it can be picked up and that left a journal to pick up from. Enter goes **into** the run +under the cursor — which says where that run is written down, and offers what there is to do +with it: ![the /epics list with the run that can be picked up marked, and what opens inside one run: its directory, over resuming it and exporting it](/demo/epics.gif) @@ -149,11 +157,13 @@ its directory, over resuming it and exporting it](/demo/epics.gif) | **resume this run** | Pick this run up, from where its journal says it got to | | **export it** | The whole run as one archive, its [trace](/user/tracing) and its session logs in it — see [Exporting a run](/user/export) | -**It is `/resume` with the run already named.** The reasons a run cannot be picked up are said -here in the same words, so the table above holds inside a run as well as at the prompt. +**It is `/resume` with the run already named.** The reasons that run cannot be picked up are +said here in the same words, so the table above holds inside a run as well as at the prompt, +from ` cannot be read back` down. -The mark in the list and that first row are one question, asked of the **flow** rather than of -the run. The weaver may have rewritten it since, so what can happen next is what it says today: +The mark in the list and that first row both ask the **flow** rather than the run — the mark +asks as well that the run left a journal. The weaver may have rewritten the flow since, so what +can happen next is what it says today: - A flow that has since dropped `resumable=True` has neither the mark nor the row, whatever the run wrote down at the time; where the row is gone, the reason stands under the list. @@ -166,10 +176,10 @@ stops a flow. ## What carrying on runs -The flow, its agents, its environments, its params and what it was asked to do all come off the -run rather than off whatever the interface happens to be set up on — an agent swapped under it -would be a different run wearing its name. The person at the prompt is not an agent anybody -chose, so a flow that talks to one talks to whoever is there now. +The flow, its agents, its environments, its params, its budget and what it was asked to do all +come off the run rather than off whatever the interface happens to be set up on — an agent +swapped under it would be a different run wearing its name. The person at the prompt is not an +agent anybody chose, so a flow that talks to one talks to whoever is there now. ## An epic is never reopened diff --git a/docs/user/security.md b/docs/user/security.md index 1128ca80..c21c7f97 100644 --- a/docs/user/security.md +++ b/docs/user/security.md @@ -50,8 +50,8 @@ So adding a flowverse trusts that git repository with this machine, exactly as i package does. Add the ones you would clone and run. `official` is always there — `chat` ships with the package, and the rest is -[humanfia/flowverse](https://github.com/humanfia/flowverse). humanize does not fetch that -repository until something wants what is in it. +[humanfia/flowverse](https://github.com/humanfia/flowverse). The interface fetches it, as it +fetches every flowverse, in the background each time it opens. ## An `hmz internal anchor` port is equivalent to a shell on that machine diff --git a/docs/user/settings.md b/docs/user/settings.md index 7ef22d5f..a00d1095 100644 --- a/docs/user/settings.md +++ b/docs/user/settings.md @@ -6,10 +6,11 @@ this when you want to see what a directory remembers, change it, or have it forg What it remembers: - the **flow** that was last run there (a flow is a directory of Python); -- for **each** flow that workspace has run: what each of its **agents** was running (an agent - is a CLI the flow runs), where its turns landed, which [account](/user/providers) it ran as - and [what it may do](/user/permissions); -- how the flow itself was [set up](/reference/tui#setting-a-flow-up); +- for **each** flow that workspace has run: what each of its **agent roles** was running (an + agent is a CLI the flow runs) and which [account](/user/providers) it ran as, and where each + of its **environment roles** was; +- how the flow itself was [set up](/reference/tui#setting-a-flow-up), and what a run of it may + spend; - whether the programs a run here starts are [profiled](#whether-a-run-here-is-profiled) as well as traced. @@ -119,9 +120,9 @@ reads two things from outside the line: the workspace's rather than the run's; - whether [reporting](/user/reporting) was answered yes. -A flow that says it [can be picked up](/user/resuming) is handed what the last run of it here -left behind — the run's own doing rather than a setting, so an unattended run of one is the -next stretch rather than the same stretch again. +With `--resume`, a flow that says it [can be picked up](/user/resuming) is handed what the last +run of it here left behind — the run's own doing rather than a setting, so an unattended run of +one is the next stretch rather than the same stretch again. ## The first time diff --git a/docs/user/steering.md b/docs/user/steering.md index b7f159cf..41eb7703 100644 --- a/docs/user/steering.md +++ b/docs/user/steering.md @@ -23,7 +23,7 @@ dropped. A held line is pinned onto the editor rather than written into the tran behind the same `❯`: ``` - assistant · claude-opus-5:high + assistant · claude/claude-opus-5:high ❯ and fix the tests too input 11.2k · output 1.1k · cache_read 0 ❯ then push 84 out/s ──────────────────────────────────────────────────────────────────────── @@ -59,9 +59,10 @@ the one the screen is showing anyway. See [Many conversations at once](/user/con Four of them take a word mid-turn — Claude Code, Codex, Kimi Code and pi. The rest were handed the whole prompt up front and have nowhere to put a second one, so what they do with a line is -answer it as the turn after. **`type(session).steers` says which before anything is said**, so a -flow that means to steer asks beforehand rather than catching a `NotImplementedError` out of a -turn already an hour in. +answer it as the turn after. **`type(session).steers` says which before anything is said**, so +code driving an agent asks beforehand rather than catching a `NotImplementedError` out of a +turn already an hour in; a flow declares `SteeringAgentMixin` on the role instead, and a CLI +that cannot steer is refused before anything runs. | Backend | What a mid-turn line does | | --- | --- | @@ -108,14 +109,15 @@ session.interject("actually, use pathlib") - On a backend that can be talked to, it raises `RuntimeError` when nothing is running to hear it. -Two related hooks, both set by the flow driving the agent: +Two related hooks, both set by whatever is driving the agent: | | | | --- | --- | | `agent.waiting` | Asked as each turn starts for anything said to this agent while no turn was open. What it returns goes into that turn. | | `agent.prompting` | Asked between turns for the next thing to say, so a flow can be a conversation rather than a loop. `None` once there will be nothing more. | -That pair is how the pin in the interface works. +`waiting` is how the pin in the interface works; what a flow is told next comes to it through +its [outworlder](/weaver/human-agent) instead. ## See also diff --git a/docs/user/stopping.md b/docs/user/stopping.md index 88777ec1..eee9b8db 100644 --- a/docs/user/stopping.md +++ b/docs/user/stopping.md @@ -86,10 +86,12 @@ rather than telling it again. **Not `/clear`.** That clears the screen and nothing else. It clears the conversation being read, not the others, and nothing that is running. -**Not choosing another flow.** `/flow` is refused while one is running, with `no choosing a -flow while a flow is running: ctrl+c twice stops it first`. A run holds the agents and -environments it was started on until it ends. Stop it first, then choose. Looking at -`/flow` and leaving without choosing changes nothing. +**Not choosing another flow.** Naming one while a flow is running — `/flow ` or a `$` +line — is refused with `a flow is running; no choosing a flow`, and `/flow` on its own opens +inside the roles of the flow that is going rather than on the flows. A run holds the agents and +environments it was started on until it ends, so what is saved there is what the next run +starts on. Stop it first, then choose. Looking at `/flow` and leaving without saving changes +nothing. **Not a question ending.** A question still up when the flow ends or is stopped ends with it. Stopping is never blocked on one. diff --git a/docs/user/tally.md b/docs/user/tally.md index 289c40b1..244616d6 100644 --- a/docs/user/tally.md +++ b/docs/user/tally.md @@ -41,8 +41,8 @@ A `+` is also there when **something was counted without its kind being said** reports a lump, a turn that spanned two models and named the kinds of neither. Those tokens went on some kind and there is nothing to say which, so every column is short by part of them. -Which kinds each backend reports is a capability like any other: `counts:cache_read` and its -four siblings say who serves each one, and `hmz.runtime.flowing.briefed()` lists them. +Which kinds each backend reports is said by its agent class: `counts`, the kinds it reports — +`CodexAgent.counts`, say, from `hmz.coganchor.agents`. ## What refreshes it, and when @@ -114,8 +114,9 @@ the fetching off for good; setting it to a path or a URL reads the list from the ## Three readings, three questions -The rest of this page is the weaver's — whoever wrote the flow. Every session and every agent -answers the same three: +The rest of this page is for whoever drives agents from Python. A flow reads what it has spent +as `session.usage` and `ctx.usage` — the time, the money and the output tokens; an agent driven +by hand, and every session of it, answers the same three: | | Answers | Moves with | | --- | --- | --- | @@ -136,7 +137,7 @@ agent.juice() And what any of those came to, in money: ```python -from hmz import prices +from hmz.coganchor import prices prices.cost(agent.spent(), agent.config.model) # dollars, or None for an unlisted model prices.price("claude-haiku-4-5-20251001") # Price(model="claude-haiku-4.5", …) @@ -195,9 +196,7 @@ billed twice. **Which kinds a backend reports is a fact about the backend, not about the turn**, and it is declared rather than guessed: a turn that spent nothing on a cache write is missing that kind -exactly as a CLI that never counts one is. `AgentBase.counts` says it, the catalogue serves it -as `counts:`, and a flow can be refused an agent whose backend never reports what it -means to steer by: +exactly as a CLI that never counts one is. `AgentBase.counts` says it: | Backend | Counts | | --- | --- | @@ -230,7 +229,7 @@ tokens were spent over, and it is the honest reading of what a run costs per hou the conversation so far, sent again at every request and mostly served out of a cache: it grows with the length of the transcript rather than with the work, so a rate counting it says how long the conversation has got — and doubles the moment a backend starts reporting what it read -back out of its cache. `session.rate()` itself is per kind, so a flow can read whichever it +back out of its cache. `session.rate()` itself is per kind, so a caller can read whichever it means. The window defaults to five minutes — `hmz.coganchor.agents.base.WINDOW`, the same window the @@ -266,7 +265,7 @@ if agent.juice(over=120) < target: A loop driving an agent from Python can do that a rung a round, to hold the agent to a target; the flow API has no way of moving an agent's effort while it runs. -A window with no turn in it reads as `0.0`. There is nothing to go on, and a flow tells that +A window with no turn in it reads as `0.0`. There is nothing to go on, and a caller tells that apart from a turn that said nothing. A backend that states a whole turn's cost **after** having said what each request in it came to @@ -284,7 +283,7 @@ each says it as the turn lands. - [Efforts](/user/efforts) — what `juice` responds to - [A turn can be cut off](/features/budgets) — the same reading, used as a cap on one turn -- [Every run has an allowance](/features/allowances) — the same reading, used as a cap on the +- [Every run has a budget](/features/allowances) — the same reading, used as a cap on the whole run, money included - [Watching a run](/user/monitor) - [Agents › What it has cost, and how fast](/reference/agents#what-it-has-cost-and-how-fast) diff --git a/docs/user/tracing.md b/docs/user/tracing.md index 257ab7da..708a6f11 100644 --- a/docs/user/tracing.md +++ b/docs/user/tracing.md @@ -91,14 +91,14 @@ Every run of a flow is one **epic**, which is a directory: ~/.humanize/epics//-/ epic.jsonl what happened, a line at a time epic._.jsonl the same, for one flow the run called - the journal what a flow that can be picked up did, for --resume + resume.jsonl what a flow that can be picked up did, for --resume profile.jsonl the programs it ran, for a run that was profiled sessions//… a link per file the backend logged that session to traces/export.trace.json the trace exporting the run gathers, replaced each time traces/.trace.json one gathered by hand afterwards, which keeps every one ``` -Not all of it every time: the journal is there for a flow that [can be picked +Not all of it every time: `resume.jsonl` is there for a flow that [can be picked up](/reference/flows#a-flow-that-can-be-picked-up), `profile.jsonl` for a directory that asked to be [profiled](#profiling-a-run), `traces/` from the first time the run is exported, and a `epic._.jsonl` for each flow the run [called](#what-a-called-flow-writes-down). @@ -121,21 +121,23 @@ yet](/demo/run.png) epic still says what it got to: ```sh -head -3 "$run"epic.jsonl +cat "$run"epic.jsonl ``` ```console -{"event":"began","at":"...","flow":"rlar","task":"...","workspace":"...","resumable":false,"agents":[{"agent":"actor",...}]} +{"event":"began","at":"...","flow":"rlar","task":"...","workspace":"...","resumable":false,"ref":"rlar:rlar","agents":[{"agent":"actor","backend":"claude","model":"claude-opus-5","effort":"high","provider":""},...],"envs":[],"params":{...},"budget":{...}} {"event":"opened","at":"...","agent":"actor","backend":"claude","provider":"local","session":"0a1b2c3d-...","name":"actor-claude@local-0a1b2c3d-...","where":"sessions/actor-claude@local-0a1b2c3d-..."} +{"event":"usage","at":"...","cost":0.61,"output_tokens":1200,"seconds":74.0} {"event":"ended","at":"...","how":"done"} ``` | `event` | Written | Carries | | --- | --- | --- | -| `began` | when the flow starts | `flow`, `task`, `workspace`, whether the flow can be picked up again and which run this one was picked up from, and one entry per agent with its id, backend, model, effort, account, what it may do, whether it could use goals and whether it was the person at the prompt | +| `began` | when the flow starts | `flow`, `task`, `workspace`, whether the flow can be picked up again (`resumable`), its canonical `ref`, which run this one was `picked_up` from where it was, one entry per agent role with its `agent`, `backend`, `model`, `effort` and `provider`, the `envs` as `-e` spells each, the `params` and the `budget` | | `opened` | each time an agent opens a session | `agent`, `backend`, `provider`, `session`, the name the run gives it and where inside the epic its links are | -| `called` | when the flow calls another flow | `flow`, `task`, and the `epic` — the record that call was written to | +| `called` | when the flow calls another flow | `flow` — its canonical ref — `task`, and the `epic` — the record that call was written to | | `returned` | when that call returns, however it ended | `flow` and the same `epic` | +| `usage` | just before `ended` | what the run spent: `cost`, `output_tokens` and `seconds` | | `ended` | when the flow stops | `how`: `done`, `failed`, or `stopped` | Each session's own logs are pointed at from `sessions//`, under a name that says whose @@ -154,7 +156,7 @@ else means following them and carrying what is behind them, which is what Claude Code's own log](/demo/run-linked.png) `/epics` is the same list at the prompt: every run of this directory, newest first, with a mark -on the ones whose flow says it can be picked up. Enter goes **into** the run under the cursor, +on the ones that can be picked up. Enter goes **into** the run under the cursor, which says where it is written down and offers two things: [export it](/user/export), which gathers a trace of that run and packs the whole of it up, and resuming it — which is [picking a run up](/user/resuming#carrying-an-older-one-on). Exporting is offered for every @@ -187,7 +189,7 @@ ls "$run"epic.*.jsonl ``` ```console -epic.jsonl epic.gen-plan_0a1b2c.jsonl +epic.jsonl epic.humanize1-gen-plan_0a1b2c.jsonl ``` The run's own record says what it called and which file to read it in: diff --git a/docs/user/troubleshooting.md b/docs/user/troubleshooting.md index 7b4bfeba..2d68c89c 100644 --- a/docs/user/troubleshooting.md +++ b/docs/user/troubleshooting.md @@ -12,7 +12,7 @@ happened. ## Starting a flow -### `no agent was given for 'reviewer'` +### `… needs an agent for 'reviewer'; give each with -a ROLE=CLI/MODEL:EFFORT` The flow declares an agent role nothing on the line filled. Give one `-a` per agent role, by the role's name: @@ -36,15 +36,16 @@ for role in flow.describe().agents: A role typed as an `Outworlder` — the person — and one typed as a `LocalEnv` — the directory the run started in — do **not** need filling. Nobody chooses what the person runs. -### ` declares no agent role 'builder'` +### ` has no agent role 'builder'; its agent roles are …` The line named a role the flow has not got — a typo, or the name another flow gives its agent. The flows each name their own: `agent` for `ralph_loop`, `actor` and `reviewer` for `rlar`. -### `the agent role 'human' is filled by the runtime, and cannot be given` +### `'human' is filled by the runtime -- whoever is outside the run -- and is not given with -a` The line named a role humanize fills itself: an `Outworlder`, which is whoever is outside the -run, or a `LocalEnv`, which is the directory the run was started in. Take it off the line. +run, or — as `… is the workspace the run is started in, and is not given with -e` — a +`LocalEnv`, which is the directory the run was started in. Take it off the line. ### `-a 'claude/claude-opus-5:high': expected =[@]/:` @@ -63,7 +64,7 @@ An `-a` is missing a part. The CLI, the model and the effort are all three requi The CLI is read from the front and the effort from after the **last** colon. A model with slashes in it, such as `kimi/kimi-code/k3:high`, is fine. -### `… needs a budget` +### `… a run is given a budget -- -b duration=...,cost=...,output_tokens=... -- and this one was given none` Every flow but `chat` is run under a budget, and `hmz exec` will not start one without it: @@ -74,7 +75,7 @@ Every flow but `chat` is run under a budget, and `hmz exec` will not start one w See [Run it unattended](/user/unattended#say-what-the-run-may-spend). -### `'worker' needs GoalCommandAgentMixin, which pi does not serve` +### `'worker' needs GoalCommandAgentMixin, which pi does not do` The role declares something that CLI cannot do — a goal, being steered, a hook only some CLIs reach. The flow is written for agents that can; pick a CLI that serves it. Which does what is in @@ -268,14 +269,13 @@ If it does not, run fewer opencode agents at once. ### `codex: this machine will not run an agent at bypass, so it runs at auto` -Not a failure: a note, said once per agent whose flow declared `bypass`. This Codex was given -requirements by somebody else — an enterprise policy that arrives with the account, or a -`requirements.toml` on a machine whose platform packages Codex — forbidding the -`danger-full-access` sandbox that [`bypass`](/user/permissions) is. Codex refuses such a call -outright, so humanize asks again a rung down, at `auto`: the same freedom, with Codex asking -before it reaches past the workspace and humanize granting what it asks. Ask for the agent at -`permission=auto` to say it yourself and skip the note. What the machine allows is its own to -say: +Not a failure: a note, said once per agent that runs at `bypass` — one whose role may write its +workdir. This Codex was given requirements by somebody else — an enterprise policy that arrives +with the account, or a `requirements.toml` on a machine whose platform packages Codex — +forbidding the `danger-full-access` sandbox that [`bypass`](/user/permissions) is. Codex refuses +such a call outright, so humanize asks again a rung down, at `auto`: the same freedom, with +Codex asking before it reaches past the workspace and humanize granting what it asks. What the +machine allows is its own to say: ```sh cat /etc/codex/requirements.toml @@ -292,15 +292,16 @@ drives the CLI you already have; it holds no API key and talks to no model provi command -v claude codex kimi pi opencode mimo zcode ``` -### `no choosing a flow while a flow is running: ctrl+c twice stops it first` +### `a flow is running; no choosing a flow` -Or `no switching flow while a flow is running`. Choosing a flow means running it, which means +A `/flow ` or a `$` line while a flow runs. Choosing a flow means running it, which means stopping whatever was running — and humanize says so rather than doing it behind your back. Press ctrl+c twice first, or type [`/stop`](/user/stopping), which is the same stop asked once. +`/flow` on its own is not refused: it opens inside the roles of the flow that is going. ### `a flow is already running` -This has the same cause, from a `/flow` that named a path. +This has the same cause: a flow was to be started while another was still running. ### `say on or off, not 'yes'` diff --git a/docs/user/tutorials/take-home.md b/docs/user/tutorials/take-home.md index 6dac6267..7eb9858f 100644 --- a/docs/user/tutorials/take-home.md +++ b/docs/user/tutorials/take-home.md @@ -124,8 +124,10 @@ the `first_chaser`, Claude Code, the second to the `second_chaser`, Codex, and r `-b` is what the run may spend — eight hours or a hundred dollars, whichever comes first — and `hmz exec` will not start a loop without one. -The first time you name a flow only the official flowverse holds, humanize fetches the [official -flowverse](/weaver/flowverses) — a git repository of flows — into `~/.humanize/flowverses/`. +`flame_chase` is in the [official flowverse](/weaver/flowverses) — a git repository of flows — +which humanize fetches into `~/.humanize/flowverses/` as `/flow` first opens in the interface. +Until it has, `hmz exec` naming one of its flows says so: open `hmz` once and press `/flow`, or +fetch it with `r` at `/flowverses`. ::: warning `flame_chase` never stops itself There is no exit condition in those lines, because "as few cycles as possible" has no end. The diff --git a/docs/user/unattended.md b/docs/user/unattended.md index b2e4e8a4..afbf5042 100644 --- a/docs/user/unattended.md +++ b/docs/user/unattended.md @@ -47,7 +47,7 @@ account, not the model. ```console $ hmz exec -f rlar -a actor=claude/claude-opus-5:max -b cost=20 "fix the build" -hmz exec: error: rlar:rlar: no agent was given for 'reviewer' +hmz exec: error: rlar needs an agent for 'reviewer'; give each with -a ROLE=CLI/MODEL:EFFORT ``` Read an agent from both ends: the CLI comes first, and the effort comes after the **last** @@ -84,8 +84,8 @@ hmz exec -f ralph_loop -a agent=claude/claude-opus-5:high -b duration=6h,cost=50 | `graceful` | `false` cuts the turn under way off the moment a limit is reached; the default lets it finish | At least one limit, and whichever is reached first stops the run. A flow calling another shares -it: what the callee spends counts against the caller too. See [Every run has an -allowance](/features/allowances). +it: what the callee spends counts against the caller too. See [Every run has a +budget](/features/allowances). ## Narrow what an agent may do @@ -168,13 +168,16 @@ Run these on purpose. Each is refused before a single turn: ```console $ hmz exec -f rlar -a actor=claude/claude-opus-5:max -b cost=20 "fix the build" -hmz exec: error: rlar:rlar: no agent was given for 'reviewer' +hmz exec: error: rlar needs an agent for 'reviewer'; give each with -a ROLE=CLI/MODEL:EFFORT $ hmz exec -f ralph_loop -a claude/claude-opus-5:high -b cost=5 "fix the build" +usage: hmz exec [-h] -f FLOW [-a ROLE=SPEC[,...]] [-e ROLE=SPEC[,...]] + [-p KEY=VALUE[,...]] [-b KEY=VALUE[,...]] [--resume] [--json] + task hmz exec: error: -a 'claude/claude-opus-5:high': expected =[@]/: $ hmz exec -f ralph_loop -a agent=claude/claude-opus-5:high "fix the build" -hmz exec: error: ralph_loop needs a budget: -b duration=…,cost=…,output_tokens=… +hmz exec: error: ralph_loop: a run is given a budget -- -b duration=...,cost=...,output_tokens=... -- and this one was given none ``` Everything that can be known before the first turn is checked before the first turn: a missing @@ -211,7 +214,7 @@ the flow will not run is refused where you wrote it. | | | | --- | --- | -| `0` | it did what it was asked | +| `0` | it did what it was asked, or spent its budget — said as `hmz exec: stopped -- …` | | `1` | it could not — no such provider, target unreachable, a turn that could not be supervised | | `2` | the command line was wrong | | `130` | interrupted | @@ -231,12 +234,13 @@ Stop it with **ctrl+c**. The interrupt reaches the whole process group, so the a process takes it too. The turn under way dies with it, what it was doing is left where it got to, and the command exits `130`. -The [epic](/user/tracing#what-a-run-writes-down) records that run as **`failed`**. `stopped` -is for a run [told to stop by hand](/user/stopping), with ctrl+c twice or `/stop` in the -interface. Nothing on a command line tells the two apart. +The [epic](/user/tracing#what-a-run-writes-down) records that run as **`stopped`**, as it +does a run [told to stop by hand](/user/stopping) with ctrl+c twice or `/stop` in the interface, +and a run its budget stopped: a run interrupted from outside is stopped, whatever the turn under +way made of it. -Either way, a flow that says it [can be picked up](/user/resuming) carries on from what that -run left behind: the same line with `--resume` picks up the newest run of that flow here, or +A flow that says it [can be picked up](/user/resuming) carries on from what that run left +behind: the same line with `--resume` picks up the newest run of that flow here, or type `/resume` in the interface. ## Checking a line before it goes into cron diff --git a/docs/weaver/branching.md b/docs/weaver/branching.md index d76b7703..3eeb1ea4 100644 --- a/docs/weaver/branching.md +++ b/docs/weaver/branching.md @@ -62,7 +62,8 @@ So the child is a conversation in every way a run counts one: | **Its own future** | turns of one are not turns of the other | It belongs to the same agent as the parent — the same CLI, model and grant, and the same -[hooks](/weaver/hooks) — and is closed with the flow that opened it, like any session. +[hooks](/weaver/hooks) — and is closed like any session: as soon as nothing holds it, or when +the flow call that opened it ends, whichever comes first. ## Use the child before the parent moves on diff --git a/specs/runtime/SPEC.md b/specs/runtime/SPEC.md index ae8aca39..21029853 100644 --- a/specs/runtime/SPEC.md +++ b/specs/runtime/SPEC.md @@ -231,13 +231,18 @@ class Runner: def run(self, task: str, *, outworlder: OutworlderDriver | None = None) -> Any: ... class Recorder: # answers to runtime/flowing's Recorder, writing the epic started: bool + def began(self, spent: Callable[[], Usage]) -> None: ... def entered(self, call: LiveCall) -> None: ... def left(self, call: LiveCall, error: BaseException | None) -> None: ... def spawned(self, call: LiveCall, role: str, session: SessionHandle, driver: AgentDriver) -> None: ... + def named(self, call: LiveCall, role: str, session: SessionHandle, + driver: AgentDriver) -> None: ... + def closed(self, session: SessionHandle) -> None: ... @property - def sessions(self) -> tuple[SessionHandle, ...]: ... - def usage(self) -> Usage: ... + def sessions(self) -> tuple[SessionHandle, ...]: ... # the ones still open + def usage(self) -> Usage: ... # the engine's reckoning of the run + def finished(self) -> Usage: ... # the same, kept once the run is over ``` ## Requirements @@ -302,8 +307,9 @@ class Recorder: # answers to runtime/flowing's Recorder, writing the epic - `arun` MUST probe every environment it was given before the flow is called, MUST run the flow over the drivers with the workspace as every `LocalEnv` role and whoever is outside the run as every `Outworlder` role -- nobody, away, where none was given -- MUST write the - run down as it goes: each flow call a record under the one that made it, each session in - the record of the call that opened it and named for its role, and what the run spent; and + run down as it goes: each flow call a record under the one that made it, saying the task it + was called with, each session in the record of the call that opened it and named for its + role, and what the run spent -- holding no session past its close to count it; and MUST close every driver it was given however the run ends. A run stopped from outside, or by its budget, MUST be written down as stopped rather than failed. - `read_line` MUST read the whole `hmz exec` line, MUST NOT load a flow to answer `--help`, diff --git a/specs/runtime/doing.md b/specs/runtime/doing.md index c590502c..052b81d5 100644 --- a/specs/runtime/doing.md +++ b/specs/runtime/doing.md @@ -64,7 +64,7 @@ class Run: usage: Usage; epic: Path | None; running: bool; raised: BaseException | None result: Any # properties @property - def agents(self) -> tuple[AgentBase, ...]: ... # behind each session opened, by role + def agents(self) -> tuple[AgentBase, ...]: ... # behind each session still open, by role def unreadable(self) -> str: ... def watch(self, listener: Listener) -> None: ... def opened(self, callback: Callable[[str, AgentBase, SessionBase], None]) -> None: ... diff --git a/specs/runtime/flowing.md b/specs/runtime/flowing.md index d325ef2b..3d190c79 100644 --- a/specs/runtime/flowing.md +++ b/specs/runtime/flowing.md @@ -88,11 +88,17 @@ class LiveCall: since: float id: int # its journal id, or 0 parent: LiveCall | None + task: str = "" + resumable: bool = False class Recorder(Protocol): + def began(self, spent: Callable[[], Usage]) -> None: ... # the run's own reckoning def entered(self, call: LiveCall) -> None: ... def left(self, call: LiveCall, error: BaseException | None) -> None: ... def spawned(self, call: LiveCall, role: str, session: SessionHandle, driver: AgentDriver) -> None: ... + def named(self, call: LiveCall, role: str, session: SessionHandle, + driver: AgentDriver) -> None: ... # a session its CLI named after it opened + def closed(self, session: SessionHandle) -> None: ... def define_flow(fn, *, agents, envs, params, name, description, hidden, resumable, caller_globals, caller_locals) -> Flow: ... def load_flow(ref: str, *, caller_globals: Mapping[str, Any]) -> Flow: ... @@ -336,6 +342,8 @@ def under() -> Path: ... JSON lines, appended to while the run goes, holding `call`, `set`, `del`, `end`, `session` and `tmp` records. A state write MUST be flushed as it is made; everything else MAY be batched, for no longer than a tenth of a second. The journal MUST be synced as the run ends. + A `session` record MUST carry the id the CLI gave the session, written once the turn that + named it reports what it spent, and no later than that turn's end. - A call's digest MUST cover the callee's canonical ref, the task, each role's agent (harness, account, model, effort, permission, skills) or environment (how it was derived from what a command line named, never a path), and the params. diff --git a/src/hmz/coganchor/agents/config.py b/src/hmz/coganchor/agents/config.py index 984746e1..c5c7c0d9 100644 --- a/src/hmz/coganchor/agents/config.py +++ b/src/hmz/coganchor/agents/config.py @@ -77,11 +77,10 @@ class Unserved(ValueError): # noqa: N818 -- what the setting is here, not what """Raised for a setting this backend has no way of carrying. Every shortfall in this layer was a bare `ValueError` with a sentence written where it - was found, which is everything a person needs and nothing a caller can act on: the one - place a flow's declaration meets a backend -- `hmz.runtime.flowing.driving.runs_at` -- - could read the sentence and re-raise it, and no more. To drop the one setting that could not be - carried and settle the rest, it has to know which one that was, and a sentence is not a - name. + was found, which is everything a person needs and nothing a caller can act on: whatever + settles a config onto a backend could read the sentence and re-raise it, and no more. To + drop the one setting that could not be carried and settle the rest, it has to know which + one that was, and a sentence is not a name. So the name travels beside the sentence, and the sentence is unchanged: this is a `ValueError` still, raised where the old one was and worded as the old one was, so every @@ -428,8 +427,8 @@ class Agents(NamedTuple): and an agent whose backend serves none of it, or a machine whose settings do not come to it, is refused before the first turn. By name rather than by feature, and by the names - everything else here already goes under: `hmz.runtime.flowing.checking.catalogue` is - where they are written down, together with which backends serve each. + everything else here already goes under: :meth:`hmz.coganchor.backends.Profile.tags`, + :mod:`hmz.coganchor.places` and the rungs :func:`rung` spells. Attributes: of_agent: What the backend filling the place has to serve, out of the agent vocabulary @@ -466,8 +465,7 @@ class Agents(NamedTuple): wrongly both ways: `Needs("isolated")` was satisfied by every backend there is, a machine capability carrying no backends and no backends meaning all of them, and `Needs(where=("anchor:hooked",))` was refused by every machine there is, no machine's - settings having ever carried one. `hmz.runtime.flowing.checking.catalogue` is where the - names are written down, each saying which half asks for it. + settings having ever carried one. Raises: TypeError: If `where` was written as one name rather than as a sequence of them. diff --git a/src/hmz/coganchor/agents/pi.py b/src/hmz/coganchor/agents/pi.py index 71ee183b..e7790047 100644 --- a/src/hmz/coganchor/agents/pi.py +++ b/src/hmz/coganchor/agents/pi.py @@ -141,9 +141,7 @@ class PiAgentConfig(AgentConfig): turn that named only the id would run on whichever Gemini that account has. Every field here is defaulted to what a bare `pi` already does, so an agent that sets none - of them starts the CLI as it ships, and each is a capability a flow may ask for before it - is handed an agent -- `hmz.runtime.flowing.checking.catalogue` reads this class and names them - `settings:`. Three things humanize imposes whatever this says, and + of them starts the CLI as it ships. Three things humanize imposes whatever this says, and each is written down where it is imposed: ``--mode rpc``, which is the transport this driver is -- a turn is a line written to a process that is already up, and steering and moving the effort mid-session are commands there rather than flags; ``--session-id``, diff --git a/src/hmz/coganchor/places.py b/src/hmz/coganchor/places.py index 41817203..436582b5 100644 --- a/src/hmz/coganchor/places.py +++ b/src/hmz/coganchor/places.py @@ -1,21 +1,16 @@ """The words for where an agent's turns land and how its commands are reached there. -One half of the capability vocabulary, kept here rather than beside the other half in -:mod:`hmz.runtime.flowing.checking` -- and the reason is the layering. A capability name is -two things at once: a fact somebody declares, and an ask a flow writes. The ask belongs above, -where flows are read and refused; the fact belongs wherever the thing it is about is written -down, -and every one of these is about a machine or about the road to one. `remote`, `isolated`, -`managed` and the platforms are what :attr:`~hmz.coganchor.machines.MachineConfig.capabilities` -answers with; `anchor:native-cli` and `anchor:supervised` are what -:attr:`~hmz.coganchor.anchor.AnchorConfig.capabilities` answers with. Each of those was -spelling its own words out, and the catalogue above was spelling them out a second time to -describe them -- one word in two places, right in whichever was read last. +The machine half of coganchor's capability vocabulary, and the reason it is here is the +layering: a capability name is a fact somebody declares, and the fact belongs wherever the +thing it is about is written down -- every one of these is about a machine or about the road +to one. `remote`, `isolated`, `managed` and the platforms are what +:attr:`~hmz.coganchor.machines.MachineConfig.capabilities` answers with; `anchor:native-cli` +and `anchor:supervised` are what :attr:`~hmz.coganchor.anchor.AnchorConfig.capabilities` +answers with. Each of those spelling its own words out would be one word in several places, +right in whichever was read last. So the words live here, where every one of their producers can reach them and where nothing -has to reach up into `hmz.runtime.flowing` to say what it serves. What stays above is the -prose: what the ask looks like, what a flow writes beside a place to reach for one, which is -a question about flows and no business of a layer that drives agents. +has to reach up into `hmz.runtime` to say what it serves. `anchor:hooked` and `anchor:preloaded` are here for the company they keep and not because a machine ever declares them. They are the two roads humanize reaches a turn down from *inside* diff --git a/src/hmz/runtime/doing/running.py b/src/hmz/runtime/doing/running.py index bc1ce8eb..1661a94d 100644 --- a/src/hmz/runtime/doing/running.py +++ b/src/hmz/runtime/doing/running.py @@ -48,7 +48,6 @@ def __init__( self._task = task self._outworlder = outworlder self._opened: list[Callable[[str, AgentBase, SessionBase], None]] = [] - self._agents: list[AgentBase] = [] self._epic: Path | None = None self._thread: threading.Thread | None = None self._loop: asyncio.AbstractEventLoop | None = None @@ -95,12 +94,17 @@ def usage(self) -> Usage: @property def agents(self) -> tuple[AgentBase, ...]: - """The coganchor agent behind each session the run has opened, oldest first. + """The coganchor agent behind each session of the run still open, oldest first. - Each is named for the role it was opened for, which is what its events say. + Each is named for the role it was opened for, which is what its events say. Only the + open ones: a run that opens a session a round for a week holds no more than one that + opened one, and a session that closed has no agent left to reach. """ - with self._lock: - return tuple(self._agents) + recorder = self._runner.recorder + if recorder is None: + return () + held = (getattr(one, "agent", None) for one in recorder.sessions) + return tuple(one for one in held if one is not None) @property def epic(self) -> Path | None: @@ -164,8 +168,6 @@ async def _main(self) -> Any: ) def _told(self, role: str, agent: AgentBase, session: SessionBase) -> None: - with self._lock: - self._agents.append(agent) for callback in tuple(self._opened): callback(role, agent, session) diff --git a/src/hmz/runtime/epic.py b/src/hmz/runtime/epic.py index 679eaca4..f3ee60d9 100644 --- a/src/hmz/runtime/epic.py +++ b/src/hmz/runtime/epic.py @@ -795,10 +795,13 @@ def links(self, only: str = "") -> None: this run has opened. """ with self._writing: - held = dict(self._sessions) + if not only: + held = dict(self._sessions) + elif only in self._sessions: + held = {only: self._sessions[only]} + else: + return for name, (backend, ident) in held.items(): - if only and name != only: - continue _link(self._at / SESSIONS / name, backend, ident) def write(self, event: str, **said: Any) -> None: diff --git a/src/hmz/runtime/flowing/engine.py b/src/hmz/runtime/flowing/engine.py index 1ded67fc..3e2ed5a7 100644 --- a/src/hmz/runtime/flowing/engine.py +++ b/src/hmz/runtime/flowing/engine.py @@ -345,7 +345,7 @@ async def __call__( if budget is not None and type(budget) is not Budget: raise TypeError(f"{self.ref}: budget={budget!r} is not a Budget") run = parent.run - node = Call(run, parent, self, depth, budget) + node = Call(run, parent, self, depth, budget, task) full = self._full # What follows is `_agent` and `_env` for the case every call in a large flow is: a # view the run handed out, of a kind already checked against this role -- written @@ -801,6 +801,7 @@ class Call: parent: The call that made it; for the call at the top of a run, the run's own. impl: The flow called; None for the run's own call above the top. depth: How many flows deep: 1 for the flow a run was started with. + task: What it was called to do. since: When it started, on the monotonic clock. deadline: When its budget's duration is spent, its own or any above it. state: What it keeps, for a resumable flow. @@ -826,6 +827,7 @@ class Call: "seqs", "since", "state", + "task", "tokens", ) @@ -836,6 +838,7 @@ def __init__( flow: FlowImpl | None, depth: int, own: Budget | None, + task: str = "", ) -> None: """A call about to start.""" self.run = run @@ -843,6 +846,7 @@ def __init__( self.impl = flow self.depth = depth self.own = own + self.task = task self.since = time.monotonic() self.deadline = _INF if parent is None else parent.deadline self.cost = 0.0 @@ -1103,15 +1107,18 @@ def record(self) -> LiveCall: record = self._record if record is None: parent = self.parent + impl = self.impl record = self._record = LiveCall( ref=self.ref, - name="" if self.impl is None else self.impl.name, + name="" if impl is None else impl.name, depth=self.depth, since=self.since, id=self.jid, parent=None if parent is None or parent.impl is None else parent.record(), + task=self.task, + resumable=impl is not None and impl.resumable, ) return record @@ -1167,6 +1174,10 @@ async def _released(res: Reversible[Releasable]) -> None: class Recorder(Protocol): """What a way in hears of a run as it goes, to write it down.""" + def began(self, spent: Callable[[], Usage]) -> None: + """The run began: `spent()` is what every turn of it has spent so far, from any thread.""" + ... + def entered(self, call: LiveCall) -> None: """A flow call started.""" ... @@ -1178,7 +1189,17 @@ def left(self, call: LiveCall, error: BaseException | None) -> None: def spawned( self, call: LiveCall, role: str, session: SessionHandle, driver: AgentDriver ) -> None: - """A flow call opened a session of an agent.""" + """A flow call opened a session of an agent, which may not be named yet.""" + ... + + def named( + self, call: LiveCall, role: str, session: SessionHandle, driver: AgentDriver + ) -> None: + """A session opened before its CLI named it has been named, as a turn of it went.""" + ... + + def closed(self, session: SessionHandle) -> None: + """A session the run opened has closed, and will spend nothing more.""" ... @@ -1193,6 +1214,8 @@ class LiveCall: since: When it started, on the monotonic clock. id: Its id in the run's journal, or 0 for a run that keeps none. parent: The call that made it, or None for the one the run was started with. + task: What it was called to do. + resumable: Whether its flow says it can be picked up again. """ ref: str @@ -1201,6 +1224,8 @@ class LiveCall: since: float id: int parent: LiveCall | None + task: str = "" + resumable: bool = False def running() -> tuple[LiveCall, ...]: @@ -1337,8 +1362,34 @@ def _drained(self) -> None: def spawned( self, node: Call, role: str, handle: SessionHandle, driver: AgentDriver + ) -> bool: + """A session was opened: told, and written down if its CLI has named it yet. + + Returns: + Whether its name is still to be written down and told: a CLI names a session as + its first turn goes, and the journal and the recorder wait for the name -- see + :meth:`named`. + """ + recorder = self.recorder + if recorder is not None: + recorder.spawned(node.record(), role, handle, driver) + if handle.id is not None: + self._noted(node, role, handle, driver) + return False + return self.journal is not None or recorder is not None + + def named( + self, node: Call, role: str, handle: SessionHandle, driver: AgentDriver ) -> None: - """A session was opened: written down, and told.""" + """A session the call `node` opened has been named by its CLI: written down, and told.""" + self._noted(node, role, handle, driver) + if self.recorder is not None: + self.recorder.named(node.record(), role, handle, driver) + + def _noted( + self, node: Call, role: str, handle: SessionHandle, driver: AgentDriver + ) -> None: + """Writes a named session into the journal, for a run that keeps one.""" if self.journal is not None: self.journal.note( { @@ -1350,8 +1401,6 @@ def spawned( "session": handle.id, } ) - if self.recorder is not None: - self.recorder.spawned(node.record(), role, handle, driver) async def close(self) -> None: """Lets go of everything the run made, however it ended.""" @@ -1608,7 +1657,9 @@ async def run_flow( local=local, recorder=recorder, ) - top = Call(run, None, None, 0, budget) + top = Call(run, None, None, 0, budget, task) + if recorder is not None: + recorder.began(lambda: top.usage) if budget.duration is not None: top.deadline = top.since + budget.duration.total_seconds() views: dict[str, AgentView] = {} diff --git a/src/hmz/runtime/flowing/fakes.py b/src/hmz/runtime/flowing/fakes.py index 19df074d..1202c436 100644 --- a/src/hmz/runtime/flowing/fakes.py +++ b/src/hmz/runtime/flowing/fakes.py @@ -282,6 +282,8 @@ def __repr__(self) -> str: @property def id(self) -> str | None: + if self.driver.names_late and not self._started: + return None return self._id @property @@ -514,6 +516,8 @@ class FakeAgentDriver: output_tokens: How many tokens each answer writes. seconds: How long each answer is reported to take; nothing actually waits. forks: Whether it can fork a session. + names_late: Whether a session says its id only once its first turn has started, as a + real CLI's does, rather than as it opens. Attributes: sessions: Every session it opened, in order. @@ -535,6 +539,7 @@ def __init__( output_tokens: int = 1, seconds: float = 0.0, forks: bool = True, + names_late: bool = False, ) -> None: self.harness = HarnessKind(harness) self.capabilities = ( @@ -550,6 +555,7 @@ def __init__( self.output_tokens = output_tokens self.seconds = seconds self.forks = forks + self.names_late = names_late self.sessions: list[FakeSession] = [] self.live = 0 self.peak = 0 diff --git a/src/hmz/runtime/flowing/journaling.py b/src/hmz/runtime/flowing/journaling.py index bdeaf725..165472a2 100644 --- a/src/hmz/runtime/flowing/journaling.py +++ b/src/hmz/runtime/flowing/journaling.py @@ -20,6 +20,11 @@ same way three times picks up the three of them in order, and a gather of identical calls picks up one apiece. +A `session` is one session a call opened, written once its CLI has named it -- as it opens +for a harness that names a session up front, as its first turn goes for one that names it +then -- so that `session` is the id the CLI logs it under. A session never named, one whose +CLI never started, is not written down. + A state write is flushed as it is made: it is what the flow will read back, and a run killed the moment after it must still have it. Everything else is batched -- written within a tenth of a second, or with the next state write, whichever comes first -- which is what keeps a run diff --git a/src/hmz/runtime/flowing/viewing.py b/src/hmz/runtime/flowing/viewing.py index e2580179..0cccb99e 100644 --- a/src/hmz/runtime/flowing/viewing.py +++ b/src/hmz/runtime/flowing/viewing.py @@ -35,6 +35,7 @@ # underscore keeps from flows rather than from them. # pyright: reportPrivateUsage=false import asyncio +import contextlib import contextvars import logging import threading @@ -169,6 +170,37 @@ def add(self, *, cost: float, output_tokens: int, duration: float) -> None: node = node.parent +class _Naming(_Sink): + """A turn's sink for a session the run is still waiting on its CLI to name. + + A CLI names a session as its first turn starts, and that turn may run for hours: the first + thing the turn is reported to have spent after that has the session written down then, + on the run's loop, rather than when the turn ends -- which a run killed mid-turn never + reaches. + """ + + __slots__ = ("_session",) + + def __init__(self, node: Call, session: SessionView) -> None: + super().__init__(node) + self._session: SessionView | None = session + + def add(self, *, cost: float, output_tokens: int, duration: float) -> None: + super().add(cost=cost, output_tokens=output_tokens, duration=duration) + with self._lock: + session = self._session + if session is None or session._handle.id is None: + return + self._session = None + run = self._node.run + if threading.get_ident() == run.thread: + session._named() + return + # A loop that is closed is a run that is over, which there is nothing left to tell. + with contextlib.suppress(RuntimeError): + run.loop.call_soon_threadsafe(session._named) + + class _Cut: """A hard deadline a turn is held to by the engine, its driver being told a sooner one. @@ -345,7 +377,7 @@ async def _opened(self, env: Env, fork_of: SessionView | None) -> SessionView: opened = Opened(session, self, env, handle) line.sessions[id(handle)] = opened node.hold(opened) - run.spawned(node, self._role, handle, self._driver) + session._unnamed = run.spawned(node, self._role, handle, self._driver) return session @overload @@ -399,7 +431,8 @@ async def run( cut = None if hard is None else _Cut(node.run.loop, handle, hard) try: said = await handle.turn( - TurnRequest(prompt, output_schema, limits), _Sink(node) + TurnRequest(prompt, output_schema, limits), + _Naming(node, taken) if taken._unnamed else _Sink(node), ) except asyncio.CancelledError: handle.interrupt() @@ -419,6 +452,9 @@ async def run( node.turning(-1) if cut is not None: cut.stop() + if taken._unnamed and handle.id is not None: + # Named by its CLI as this turn went, and nothing it spent said so sooner. + taken._named() # A fork is cut by now, and the session it was cut from may go. taken._parent = None failed = taken._error @@ -611,6 +647,7 @@ class SessionView: "_handle", "_line", "_parent", + "_unnamed", ) def __init__( @@ -628,6 +665,8 @@ def __init__( self._closed = False self._error: Exception | None = None self._parent: SessionView | None = None + #: Whether the run's journal or recorder is still waiting on its CLI to name it. + self._unnamed = False def __repr__(self) -> str: return f"" @@ -649,6 +688,22 @@ def id(self) -> str | None: """The CLI's own id for the conversation, or None before it has said one.""" return self._handle.id + def _named(self) -> None: + """Its CLI has named it: written down and told, against the call that opened it. + + Once, however many times it is asked, and never raising: a journal that cannot be + written to is a run that can no longer be picked up, not a turn that failed. + """ + if not self._unnamed: + return + self._unnamed = False + opener: AgentView = self._agent # pyright: ignore[reportAssignmentType] + node = opener._node + try: + node.run.named(node, opener._role, self._handle, opener._driver) + except Exception: + log.exception("writing down the session of %s failed", opener._role) + def _failed(self, error: Exception) -> None: """A hook of this session raised: the turn under way stops, and raises it.""" if self._error is None: @@ -749,9 +804,13 @@ async def _shut(self) -> None: except Exception: log.exception("closing %r failed", self) finally: - line = self.agent._line + agent = self.agent + line = agent._line if line is not None: line.sessions.pop(id(handle), None) + recorder = agent._node.run.recorder + if recorder is not None: + recorder.closed(handle) def _settled(self, task: asyncio.Task[None]) -> None: """Its close is over, and the call has nothing of it left to release.""" diff --git a/src/hmz/runtime/runner.py b/src/hmz/runtime/runner.py index 0c2c891a..0c42bcd7 100644 --- a/src/hmz/runtime/runner.py +++ b/src/hmz/runtime/runner.py @@ -26,9 +26,13 @@ import contextlib import json import math +import time +import weakref from pathlib import Path from typing import TYPE_CHECKING, Any, NamedTuple, cast +from . import telemetry + if TYPE_CHECKING: import os from collections.abc import Callable, Iterable, Mapping @@ -586,6 +590,7 @@ async def arun( ) recorder = Recorder(epic, opened) self._recorder = recorder + _GOING.add(self) try: with epic: if started is not None: @@ -614,7 +619,7 @@ async def arun( epic.stopped() raise finally: - usage = recorder.usage() + usage = recorder.finished() epic.write( "usage", cost=usage.cost, @@ -622,6 +627,7 @@ async def arun( seconds=usage.duration.total_seconds(), ) finally: + _GOING.discard(self) await asyncio.shield(self._closed(local)) async def aclose(self) -> None: @@ -653,7 +659,12 @@ class Recorder: """What writes a run down as the engine runs it: a record per flow call, and each session. Answers to :class:`hmz.runtime.flowing.engine.Recorder`. Every call is told on the loop - the run is on. + the run is on; what the run has spent is read from any thread. + + What the run has spent is the engine's own reckoning of it -- every turn of every session, + as its budget is held to -- rather than a sum over the sessions, so the recorder holds the + sessions still open and nothing of the ones that closed: a loop that opens a session a + round for a week holds as much as one that opened one. Attributes: started: Whether the flow the run was started with has been called, which is what @@ -662,12 +673,21 @@ class Recorder: def __init__(self, epic: Epic, opened: Opened | None = None) -> None: """Holds the epic to write into, and what to tell of each session opened.""" + import threading + self._epic = epic self._opened = opened self._records: dict[int, Epic] = {} - self._sessions: list[SessionHandle] = [] + self._lock = threading.Lock() + self._live: dict[int, SessionHandle] = {} + self._spent: Callable[[], Usage] | None = None + self._final: Usage | None = None self.started = False + def began(self, spent: Callable[[], Usage]) -> None: + """The run began, and `spent()` is what it has spent so far.""" + self._spent = spent + def entered(self, call: LiveCall) -> None: """A flow call started: the run's own, or one written into a record of its own.""" self.started = True @@ -675,7 +695,9 @@ def entered(self, call: LiveCall) -> None: self._records[id(call)] = self._epic return above = self._records.get(id(call.parent), self._epic) - self._records[id(call)] = above.called(call.ref) + self._records[id(call)] = above.called( + call.ref, call.task, resumable=call.resumable + ) def left(self, call: LiveCall, error: BaseException | None) -> None: """A flow call ended, which closes its record.""" @@ -697,15 +719,15 @@ def spawned( self, call: LiveCall, role: str, session: SessionHandle, driver: AgentDriver ) -> None: """A flow call opened a session: named for its role, and written into its record.""" - self._sessions.append(session) + with self._lock: + self._live[id(session)] = session record = self._records.get(id(call), self._epic) agent: AgentBase | None = getattr(session, "agent", None) if agent is None: - # A driver with no coganchor agent behind it -- a fake -- names its session as - # it opens it, and is written down then. - record.session( - role, str(driver.harness), driver.provider, session.id or "?" - ) + # A driver with no coganchor agent behind it -- a fake -- is written down here, + # once it has said what it calls the session: now, or when it is `named`. + if session.id is not None: + record.session(role, str(driver.harness), driver.provider, session.id) return # Named for its role, and written down in the record of the call that opened it # once its CLI has said what it calls the conversation, which is its first turn. @@ -715,30 +737,108 @@ def spawned( if self._opened is not None and conversation is not None: self._opened(role, agent, conversation) + def named( + self, call: LiveCall, role: str, session: SessionHandle, driver: AgentDriver + ) -> None: + """A session was named as a turn of it went: written into its record, if it was not. + + A coganchor agent writes its own session down as its CLI names it, into the record + :meth:`spawned` handed it; this is for a driver with none behind it. + """ + if getattr(session, "agent", None) is not None or session.id is None: + return + record = self._records.get(id(call), self._epic) + record.session(role, str(driver.harness), driver.provider, session.id) + + def closed(self, session: SessionHandle) -> None: + """A session closed, and is let go of: what it spent is already the run's.""" + with self._lock: + self._live.pop(id(session), None) + @property def sessions(self) -> tuple[SessionHandle, ...]: - """Every session the run has opened, oldest first.""" - return tuple(self._sessions) + """Every session of the run still open, oldest first.""" + with self._lock: + return tuple(self._live.values()) def usage(self) -> Usage: """Everything the run's sessions have spent, up to the moment it is read.""" - import datetime - from hmz.flows import Usage - cost = 0.0 - tokens = 0 - seconds = 0.0 - for session in tuple(self._sessions): - said = session.usage - cost += said.cost - tokens += said.output_tokens - seconds += said.duration.total_seconds() - return Usage( - duration=datetime.timedelta(seconds=seconds), - cost=cost, - output_tokens=tokens, - ) + final, spent = self._final, self._spent + if final is not None: + return final + return Usage() if spent is None else spent() + + def finished(self) -> Usage: + """The run is over: what it spent, kept as it stands, and the run let go of.""" + final = self._final = self.usage() + self._spent = None + return final + + +#: The runs going now in this process, for a report of a failure to say what was running. +_GOING: weakref.WeakSet[Runner] = weakref.WeakSet() + +#: The most flow calls a report lists: a flow of ten thousand calls is described by its +#: oldest, and by how many there were. +_LISTED = 64 + + +def _about() -> dict[str, object]: + """What was running when a failure was reported: the flows, and what each role ran. + + Registered with :mod:`telemetry` as `flow`, and asked only if a report is ever made. + Names and settings, never a task or a path: which flow, how deep, for how long, and each + agent role's CLI, model, effort, account by name, what it may do and the skills it carries. + """ + from hmz.runtime.flowing import running + + calls = running() + now = time.monotonic() + agents: list[dict[str, object]] = [] + for runner in tuple(_GOING): + for role, driver in runner.agents.items(): + declared = runner.declaration.agent(role) + may = ( + "" + if declared is None + else " ".join( + f"{scope}={getattr(declared.permission, scope)}" + for scope in ("local", "user", "system", "online") + ) + ) + agents.append( + { + "flow": runner.impl.ref, + "called": role, + "cli": str(driver.harness), + "model": driver.model, + "effort": driver.effort, + "account": driver.provider or "as this machine is signed in", + "may": may, + "skills": list(declared.skills) if declared is not None else [], + } + ) + return { + "flow": next((one.ref for one in calls if one.parent is None), ""), + "calls": len(calls), + "running": [ + { + "flow": one.ref, + "deep": one.depth, + "under": one.parent.ref if one.parent is not None else "", + "for": round(now - one.since), + } + for one in calls[:_LISTED] + ], + "agents": agents, + } + + +# Registered once, as the module is loaded: what it answers is what is running at the moment +# of a report, rather than anything one run holds. +telemetry.about("flow", _about) def _agent_drivers( diff --git a/tests/integration/runtime/test_epics.py b/tests/integration/runtime/test_epics.py index aa5c4f06..022b823e 100644 --- a/tests/integration/runtime/test_epics.py +++ b/tests/integration/runtime/test_epics.py @@ -344,6 +344,32 @@ def test_the_logs_of_a_session_are_linked_into_the_epic_that_opened_it( assert linked(epic) == {one.name: [str(log)]} +def test_a_resumable_run_journals_each_session_by_the_id_its_cli_gave_it( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """A CLI names a session as its first turn goes, and the journal says that name.""" + standing_in(tmp_path, monkeypatch) + monkeypatch.chdir(tmp_path) + written( + tmp_path, + "flow", + ONE.replace("params=FlowParams)", "params=FlowParams, resumable=True)"), + ) + + Runner(tmp_path / "flow", agents={"builder": AGENT}, budget=BUDGET).run(TASK) + + (epic,) = epics() + (one,) = sessions(epic) + journaled = [ + json.loads(line) + for line in (epic / "resume.jsonl").read_text().splitlines() + if '"t":"session"' in line + ] + assert [(said["role"], said["harness"], said["session"]) for said in journaled] == [ + ("builder", "claude", one.ident) + ] + + def test_a_log_written_after_the_last_turn_is_linked_when_the_run_ends( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: @@ -418,12 +444,14 @@ def test_flows_calling_flows_read_back_as_the_tree_they_ran_in( (epic,) = epics() calls = tree(epic) - assert sorted((one.flow, len(one.calls), one.how) for one in calls) == [ - ("flow:branch", 0, "done"), - ("flow:branch", 1, "done"), + assert sorted((one.flow, one.task, len(one.calls), one.how) for one in calls) == [ + ("flow:branch", "left", 1, "done"), + ("flow:branch", "right", 0, "done"), ] (deeper,) = [one for one in calls if one.calls] - assert [(one.flow, one.how) for one in deeper.calls] == [("flow:branch", "done")] + assert [(one.flow, one.task, one.how) for one in deeper.calls] == [ + ("flow:branch", "left", "done") + ] # Every session in the record of the call that opened it, and none in the run's own. records = {one.record for one in sessions(epic)} assert JOURNAL not in records diff --git a/tests/integration/runtime/test_reported.py b/tests/integration/runtime/test_reported.py new file mode 100644 index 00000000..06e5a42d --- /dev/null +++ b/tests/integration/runtime/test_reported.py @@ -0,0 +1,96 @@ +"""What a report of humanize's own failure says of the run that was going when it happened. + +`telemetry.SENT` promises a report says which flow was running and what each of its agents was +set up to run, and nothing a person typed: this is that promise held to a run on the engine's +fakes, a flow calling a flow, asked mid-run exactly as a report being made would ask it. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Any, cast + +from hmz.runtime import telemetry +from hmz.runtime.flowing.fakes import FakeAgentDriver +from hmz.runtime.runner import Runner +from tests.stubs import written + +if TYPE_CHECKING: + from pathlib import Path + + import pytest + +#: A flow that calls another, which reads what a report would say while both are going. +FLOW = """ +from hmz.flows import ( + Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, Permission, + PermissionKind, flow, +) +from hmz.runtime import telemetry + +SEEN = [] + + +class Reader(Agent): + _permission = Permission(online=PermissionKind.ALL) + _skills = () + + +class Agents(AgentCollection): + reader: Reader + + +class Envs(EnvCollection): + here: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams, hidden=True) +async def inner(task, *, agents, envs, params, ctx): + SEEN.append(telemetry.held()["flow"]) + + +@flow(agents=Agents, envs=Envs, params=FlowParams, name="flow") +async def outer(task, *, agents, envs, params, ctx): + await inner("a secret task", agents=agents, envs=envs, params=params) + return SEEN +""" + + +def test_a_report_says_which_flows_were_going_and_what_each_role_ran( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.chdir(tmp_path) + written(tmp_path, "flow", FLOW) + driver = FakeAgentDriver(model="m", effort="high", provider="work") + + (said,) = Runner( + tmp_path / "flow", agents={"reader": driver}, budget={"cost": 1} + ).run("a secret task") + + about = cast("dict[str, Any]", said) + assert about["flow"] == "flow:flow" + assert about["calls"] == 2 + assert [(one["flow"], one["deep"], one["under"]) for one in about["running"]] == [ + ("flow:flow", 1, ""), + ("flow:inner", 2, "flow:flow"), + ] + assert about["agents"] == [ + { + "flow": "flow:flow", + "called": "reader", + "cli": "claude", + "model": "m", + "effort": "high", + "account": "work", + "may": "local=all user=read system=read online=all", + "skills": [], + } + ] + # What the flow was asked to do is nowhere in it. + assert "secret" not in repr(about) + # And once the run is over, nothing of it is left to say. + assert telemetry.held()["flow"] == { + "flow": "", + "calls": 0, + "running": [], + "agents": [], + } diff --git a/tests/unit/flows/test_engine_calls.py b/tests/unit/flows/test_engine_calls.py index d2af3f35..454593f6 100644 --- a/tests/unit/flows/test_engine_calls.py +++ b/tests/unit/flows/test_engine_calls.py @@ -58,6 +58,9 @@ ) if TYPE_CHECKING: + from collections.abc import Callable + + from hmz.flows import Usage from hmz.runtime.flowing.spi import AgentDriver, SessionHandle @@ -937,8 +940,11 @@ class Heard: def __init__(self) -> None: self.said: list[tuple[Any, ...]] = [] + def began(self, spent: Callable[[], Usage]) -> None: + self.said.append(("began", spent().output_tokens)) + def entered(self, call: LiveCall) -> None: - self.said.append(("entered", call.name, call.depth)) + self.said.append(("entered", call.name, call.depth, call.task)) def left(self, call: LiveCall, error: BaseException | None) -> None: self.said.append(("left", call.name, type(error).__name__ if error else None)) @@ -948,23 +954,35 @@ def spawned( ) -> None: self.said.append(("spawned", call.name, role, session.id is not None)) + def named( + self, call: LiveCall, role: str, session: SessionHandle, driver: AgentDriver + ) -> None: + self.said.append(("named", call.name, role)) + + def closed(self, session: SessionHandle) -> None: + self.said.append(("closed", session.id is not None)) + async def test_a_recorder_hears_every_call_and_session() -> None: @flow(agents=Solo, envs=Place, params=Nothing) async def recorded( task: str, *, agents: Solo, envs: Place, params: Nothing, ctx: FlowContext ) -> None: - await agents["agent"].spawn(env=envs["env"]) + session = await agents["agent"].spawn(env=envs["env"]) with pytest.raises(BoomError): - await boom(task, agents={}, envs={}, params=Depth()) + await boom("under", agents={}, envs={}, params=Depth()) + del session + await asyncio.sleep(0) heard = Heard() - await run_fake(recorded, recorder=heard) + await run_fake(recorded, "over", recorder=heard) assert heard.said == [ - ("entered", "recorded", 1), + ("began", 0), + ("entered", "recorded", 1, "over"), ("spawned", "recorded", "agent", True), - ("entered", "boom", 2), + ("entered", "boom", 2, "under"), ("left", "boom", "BoomError"), + ("closed", True), ("left", "recorded", None), ] diff --git a/tests/unit/flows/test_engine_resume.py b/tests/unit/flows/test_engine_resume.py index a228770f..fabc395d 100644 --- a/tests/unit/flows/test_engine_resume.py +++ b/tests/unit/flows/test_engine_resume.py @@ -26,6 +26,8 @@ FlowContext, FlowParams, StateNotSerializable, + StopHookParams, + StopHookResult, TemporaryClonedDirEnvMixin, flow, ) @@ -322,6 +324,80 @@ async def recorded( assert all(one["ok"] for one in ends) +async def test_a_session_is_written_down_once_its_cli_has_named_it( + tmp_path: Path, +) -> None: + """A CLI names a session as its first turn goes: the record waits for the name.""" + driver = FakeAgentDriver(names_late=True) + journal = tmp_path / "run.jsonl" + unnamed: list[bool] = [] + + @flow(agents=Solo, envs=Place, params=Step, resumable=True) + async def named( + task: str, *, agents: Solo, envs: Place, params: Step, ctx: FlowContext + ) -> None: + agent = agents["agent"] + session = await agent.spawn(env=envs["env"]) + idle = await agent.spawn(env=envs["env"]) + unnamed.append(driver.sessions[0].id is None) + # Twice, and through an agent derived from the one that opened it: written down once. + await agent.run(task, session=session) + await agent.derive().run(task, session=session) + del idle + + await _run(named, journal, resume=False, agents={"agent": driver}) + used, idle = driver.sessions + assert unnamed == [True] + assert used.id is not None + assert idle.id is None + records = _records(journal) + (top,) = (one for one in records if one["t"] == "call") + sessions = [one for one in records if one["t"] == "session"] + # One record, for the session that took a turn, against the call that opened it, and + # none for the one that never did: there is no conversation of it to find. + assert sessions == [ + { + "t": "session", + "id": top["id"], + "role": "agent", + "harness": "claude", + "model": "fake", + "session": used.id, + } + ] + + +async def test_a_session_named_mid_turn_is_written_down_before_the_turn_ends( + tmp_path: Path, +) -> None: + """A first turn may run for hours: what it is reported to have spent says it was named.""" + journal = tmp_path / "run.jsonl" + seen: list[list[str]] = [] + + @flow(agents=Solo, envs=Place, params=Step, resumable=True) + async def long( + task: str, *, agents: Solo, envs: Place, params: Step, ctx: FlowContext + ) -> None: + state = ctx.state + assert state is not None + + async def stopping(hooked: StopHookParams) -> StopHookResult: + del hooked + state["flushed"] = True # a state write writes whatever is waiting + seen.append([one["t"] for one in _records(journal)]) + return StopHookResult() + + agent = agents["agent"] + agent.on_stop(stopping) + session = await agent.spawn(env=envs["env"]) + await agent.run(task, session=session) + + await _run( + long, journal, resume=False, agents={"agent": FakeAgentDriver(names_late=True)} + ) + assert seen == [["journal", "call", "session", "set"]] + + async def test_a_state_write_is_on_disk_before_the_call_goes_on(tmp_path: Path) -> None: journal = tmp_path / "run.jsonl" seen: list[list[dict[str, Any]]] = [] diff --git a/tests/unit/runtime/test_recorder.py b/tests/unit/runtime/test_recorder.py new file mode 100644 index 00000000..26fdee8e --- /dev/null +++ b/tests/unit/runtime/test_recorder.py @@ -0,0 +1,205 @@ +"""What writes a run down as the engine runs it, and what it holds on to while it does. + +A run is written into its epic call by call and session by session, and what the run has spent +is read off the recorder while it goes -- so the recorder is with the run for as long as the +run is, which for a loop meant to go for a week is every session that loop ever opened. It +holds the ones still open and a total of the rest, and nothing more. +""" + +from __future__ import annotations + +import sys +from typing import cast + +import pytest + +from hmz.flows import ( + Agent, + AgentCollection, + EnvCollection, + FlowContext, + FlowParams, + HarnessKind, + LocalEnv, + flow, +) +from hmz.runtime.epic import Epic, sessions, tree +from hmz.runtime.flowing.fakes import FakeAgentDriver, run_fake +from hmz.runtime.runner import Recorder + + +class Solo(AgentCollection): + agent: Agent + + +class Here(EnvCollection): + here: LocalEnv + + +class Rounds(FlowParams): + rounds: int = 1 + + +#: How many sessions the long loop opens. +SESSIONS = 10_000 + + +@flow(agents=Solo, envs=Here, params=Rounds) +async def rounds( + task: str, *, agents: Solo, envs: Here, params: Rounds, ctx: FlowContext +) -> None: + for _ in range(params.rounds): + session = await agents["agent"].spawn(env=envs["here"]) + await agents["agent"].run(task, session=session) + + +@flow(agents=Solo, envs=Here, params=Rounds) +async def calls( + task: str, *, agents: Solo, envs: Here, params: Rounds, ctx: FlowContext +) -> None: + await rounds(f"{task}, once", agents=agents, envs=envs, params=Rounds()) + + +def _held_by(recorder: Recorder) -> int: + """The bytes the recorder's own attributes and containers take, in bytes. + + Into its dicts, lists, tuples and sets and no further: an epic or a session it holds is + counted as the one object it holds, not as everything that object holds in turn. + """ + seen: set[int] = set() + reach: list[object] = [*vars(recorder).values()] + total = sys.getsizeof(recorder) + while reach: + one = reach.pop() + if id(one) in seen: + continue + seen.add(id(one)) + total += sys.getsizeof(one) + if isinstance(one, dict): + reach.extend(cast("dict[object, object]", one).keys()) + reach.extend(cast("dict[object, object]", one).values()) + elif isinstance(one, list | tuple | set): + reach.extend(cast("list[object]", one)) + return total + + +async def test_a_loop_of_ten_thousand_sessions_leaves_the_recorder_holding_none() -> ( + None +): + """Each session let go of as it closes, what it spent kept as a sum.""" + epic = Epic("rounds", "go") + recorder = Recorder(epic) + open_at: list[int] = [] + grown: list[int] = [] + + @flow(agents=Solo, envs=Here, params=Rounds) + async def watched( + task: str, *, agents: Solo, envs: Here, params: Rounds, ctx: FlowContext + ) -> None: + agent = agents["agent"] + for n in range(params.rounds): + session = await agent.spawn(env=envs["here"]) + await agent.run(task, session=session) + open_at.append(len(recorder.sessions)) + if n in (SESSIONS // 10, SESSIONS - 1): + grown.append(_held_by(recorder)) + + # An ACP agent, whose sessions nobody logs: the epic has nothing to link each one to. + driver = FakeAgentDriver(HarnessKind.ACP, cost=0.001, output_tokens=3) + with epic: + await run_fake( + watched, + "go", + agents={"agent": driver}, + params={"rounds": SESSIONS}, + recorder=recorder, + ) + + assert max(open_at) <= 2, "the recorder held sessions that had closed" + assert recorder.sessions == () + early, late = grown + assert late - early < 1024, f"{late - early} bytes more after 9,000 more sessions" + spent = recorder.usage() + assert spent.output_tokens == 3 * SESSIONS + assert spent.cost == pytest.approx(0.001 * SESSIONS) + + +async def test_what_a_run_spent_counts_its_open_sessions_and_its_closed_ones() -> None: + """Read while a session is open, and again once it has closed: the same sum.""" + epic = Epic("rounds", "go") + recorder = Recorder(epic) + seen: list[tuple[int, int]] = [] + + @flow(agents=Solo, envs=Here, params=Rounds) + async def spends( + task: str, *, agents: Solo, envs: Here, params: Rounds, ctx: FlowContext + ) -> None: + agent = agents["agent"] + first = await agent.spawn(env=envs["here"]) + await agent.run(task, session=first) + del first + kept = await agent.spawn(env=envs["here"]) + await agent.run(task, session=kept) + seen.append((len(recorder.sessions), recorder.usage().output_tokens)) + + with epic: + await run_fake( + spends, + "go", + agents={"agent": FakeAgentDriver(output_tokens=5)}, + recorder=recorder, + ) + + assert seen == [(1, 10)] + assert recorder.sessions == () + assert recorder.usage().output_tokens == 10 + + +async def test_a_session_named_late_is_written_down_by_its_name() -> None: + """Once its CLI has named it, rather than as it opened with a name it did not have.""" + epic = Epic("rounds", "go") + driver = FakeAgentDriver(names_late=True) + with epic: + await run_fake(rounds, "go", agents={"agent": driver}, recorder=Recorder(epic)) + + (session,) = sessions(epic.path) + assert session.ident == driver.sessions[0].id + + +async def test_what_a_run_spent_is_what_its_budget_was_held_to() -> None: + """The engine's own reckoning, which a session closed mid-turn goes on adding to.""" + epic = Epic("rounds", "go") + recorder = Recorder(epic) + reckoned: list[int] = [] + + @flow(agents=Solo, envs=Here, params=Rounds) + async def spends( + task: str, *, agents: Solo, envs: Here, params: Rounds, ctx: FlowContext + ) -> None: + await rounds(task, agents=agents, envs=envs, params=Rounds(rounds=3)) + reckoned.append(ctx.usage.output_tokens) + + with epic: + await run_fake( + spends, + "go", + agents={"agent": FakeAgentDriver(output_tokens=7)}, + recorder=recorder, + ) + + assert reckoned == [21] + assert recorder.finished().output_tokens == 21 + assert recorder.usage().output_tokens == 21 + + +async def test_a_flow_called_is_written_down_with_the_task_it_was_called_with() -> None: + epic = Epic("calls", "go") + with epic: + await run_fake(calls, "go", recorder=Recorder(epic)) + + (one,) = tree(epic.path) + assert (one.flow.rpartition(":")[2], one.task, one.how) == ( + "rounds", + "go, once", + "done", + )