diff --git a/pyproject.toml b/pyproject.toml index b46a443d..a6d34338 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -232,11 +232,6 @@ ignore = [ # both modules say so where they reach for one -- `find` in one, `_own` in the other. "src/hmz/runtime/flowing/finding.py" = ["PTH"] "src/hmz/runtime/flowing/verses.py" = ["PTH"] -# A flow is a directory whose `__init__.py` holds its entry points, and that module's -# docstring is about them -- how the flow loops, what each agent is for, and the line -# that starts it. What a flow prints is what it says: the interface captures everything -# printed under it into the transcript, so a warning a flow has to give is a `print`. -"src/hmz/_legacy_flows/builtin/**" = ["D103", "T201"] # The exceptions a flow catches are named for what happened -- `FlowNotFound`, # `BudgetExceeded`, `TempCloneBusy` -- as the flow API's spec names them, rather than each # given an `Error` suffix that would say only that they are exceptions. diff --git a/specs/SPEC.md b/specs/SPEC.md index 06801203..8fe3bbe0 100644 --- a/specs/SPEC.md +++ b/specs/SPEC.md @@ -52,7 +52,8 @@ def home() -> pathlib.Path: ... the protocols a flow's agents, environments, sessions and context answer to, and the values a flow writes or catches. The objects a flow is handed MUST be the runtime's, answering to those protocols structurally. Everything humanize does *to* a flow MUST be `runtime/flowing` - instead. + instead. The flows humanize ships MUST be kept in `flows/builtin`, written against `flows` + like any other flow and importing nothing else. - `cli`, `daemon` and `tui` MUST each be a way of reaching the runtime's one object rather than a second copy of what it does. Anything two of them would otherwise each have written MUST be written in `runtime` instead, so that a thing which can be done one way can be done @@ -68,9 +69,6 @@ def home() -> pathlib.Path: ... `Outworlder.new` to `runtime/flowing`, importing it inside the call and never at import. `flows` MUST import nothing else of humanize, and MUST be checked to do so rather than taken on trust. -- `_legacy_flows` is the previous flow API, kept whole until every way in has moved to - `flows`. It MAY go on naming `coganchor`, and pairing with `runtime/flowing`, as it did - under the old name; nothing new MUST import it, and it MUST go when nothing does. - `cli` MUST reach `runtime` by name. `tui` MUST reach it through `daemon`, and `daemon` MUST offer it. `sdk` MUST be named by no layer. - `coganchor/serve` — the half that ships to a target of any architecture — MUST name the diff --git a/specs/cli.md b/specs/cli.md index 96020949..ecd0e27b 100644 --- a/specs/cli.md +++ b/specs/cli.md @@ -9,9 +9,13 @@ decides. ```shell hmz [ [...]] | hmz --version | hmz --help # no command: the terminal interface -hmz exec -f|--flow [/][:] -a|--agent [,...] [-a ...] - [-c|--config ] [--json] - := [=][@]/: +hmz exec -f|--flow [-a|--agents [,...]]... [-e|--envs [,...]]... + [-p|--params =[,...]]... [-b|--budget [,...]]... + [--resume] [--json] + := [/][:] | | git+[@]#[:] + := =[@]/: + := =@/ + := duration= | cost= | output_tokens= | graceful= hmz internal [...] hmz internal anchor [] [...] hmz internal anchor serve --export [:] [--export ...] @@ -52,8 +56,7 @@ class Out: # a run written for a person, or as NDJSON for a program; a contex class Shown: # the agents' own events, drawn as one run; a context manager def __init__(self, out: Out) -> None: ... - def watches(self, agents: Iterable[AgentBase]) -> None: ... - def heard( + def heard( # handed to `Run.watch`, which every session of a run is watched through self, agent: AgentBase, session: SessionBase | None, event: Event ) -> None: ... @@ -89,23 +92,26 @@ def tools(argv: list[str]) -> int: ... carrying objects alone -- and MUST NOT carry what only a terminal needed. - MUST say nothing to a program that a line for a person would not: an account is its variable names and never their values, a flowverse its URL with any secret taken out. -- `hmz exec` MUST take every `-a` on the line as one list of agents in the order written, however it - was broken up, and two agents of one spelling MUST be two agents. -- A `` MAY name the place it fills; naming MUST be all or nothing, and a name the flow does not - declare, one given twice, a place left unfilled, or naming places to a flow that declared a plain - tuple MUST each be a usage error before any agent has run -- as MUST a flow that is not there, has - no entry point, or drives a different number of agents than were given. -- MUST refuse a written-out agent -- `cli=`, `model=`, `effort=`, `provider=`, `service_tier=`, - `config.=` -- and MUST refuse `permission=` and `web_search=` saying where they are said - instead. +- `hmz exec` MUST take every `-a`, `-e`, `-p` and `-b` on the line as one list apiece, however it + was broken up, and two roles given one spelling MUST be two agents. +- Every `` and `` MUST name the role it fills. A role the flow does not declare, one + given twice, a required role left unfilled, a role the runtime fills -- an `Outworlder`, a + `LocalEnv` -- named at all, a param the flow does not take or cannot read, a spec that cannot be + read, an agent whose harness is not the one its role names or does not serve what its role asks, + and a line with no `-b` -- for every flow but `chat`, which runs under `Budget(cost=inf)` -- MUST + each be a usage error before any agent has started, as MUST a flow that is not there or will not + load, and `--resume` of a flow that cannot be picked up or has no run to pick up. +- `--resume` MUST pick up the newest run of that flow in this workspace that can be picked up; + without it every run MUST start from the top. - MUST read `` from the front and `` from after the last colon so that a model's own punctuation stays the model's, and MUST NOT restate here which backends exist. +- A run started from a command line MUST have nobody outside it: its outworlder is away. - MUST draw a run from the agents' own event stream rather than from each backend's own progress and MUST NOT show both, saying which agent is taking a turn and in which conversation, what it said, what it ran, what it started, and what a turn cost -- in money as well as tokens, and tokens alone for a model nobody prices -- with something going on moving while a terminal is reading. -- MUST say, without asking, when nothing will stop the run, when a cap cannot be read, and what a flow - declared that its agent could not be told. +- MUST say, without asking, when a cap cannot be read -- a cost cap over a model nobody prices -- and + MUST say a run its budget stopped in a line rather than as a failure. - `hmz internal anchor` MUST load `coganchor` and nothing else of humanize, the door included. - `hmz internal cred` MUST exit with the program's own status, MUST refuse a line naming nothing to answer or no program to run, and MUST NOT fall back to running unsupervised. diff --git a/specs/runtime/SPEC.md b/specs/runtime/SPEC.md index 143293f5..ae8aca39 100644 --- a/specs/runtime/SPEC.md +++ b/specs/runtime/SPEC.md @@ -1,15 +1,16 @@ # `runtime` -What a run is: finding the flow, handing it the agents it declared, writing the run down as it -happens, remembering what a workspace was set up with, and reading the whole of it back -afterwards. It drives no coding agent itself. Its subpackages have specs of their own: +What a run is: finding the flow, handing it a driver for every role it declared, writing the +run down as it happens, remembering what a workspace was set up with, and reading the whole of +it back afterwards. It drives no coding agent itself. Its subpackages have specs of their own: [doing](doing.md), [flowing](flowing.md), [tracing](tracing.md). ## API ```python # __init__.py -- each name costs only the module it is written in, fetched when it is named -__all__ = ["Accounts", "Epics", "Fallbacks", "Flows", "Flowverses", "Hmz", "Run"] # doing.md +__all__ = ["Accounts", "Epics", "Fallbacks", "Flows", "Flowverses", "Hmz", "Refused", "Run"] +# doing.md, and `Refused` from runner.py # settings.py -- what humanize remembers, per workspace and per machine class Settings: @@ -22,18 +23,17 @@ class Settings: def profiling(self) -> bool: ... def profiles(self, *, on: bool) -> None: ... def answers(self, *, enable_sentry: bool) -> None: ... - def agents( - self, flow: str, goal_defaults: Sequence[bool] | None = None - ) -> list[Runs]: ... + def agents(self, flow: str) -> dict[str, Runs]: ... # by role + def envs(self, flow: str) -> dict[str, str]: ... # by role, as `-e` spells one def flows(self) -> dict[str, Any]: ... - def config(self, flow: str) -> dict[str, Any]: ... - def budget(self, flow: str) -> dict[str, Any]: ... + def params(self, flow: str) -> dict[str, Any]: ... + def budget(self, flow: str) -> dict[str, Any]: ... # a Budget, as JSON def remember( self, flow: str, - names: tuple[str, ...], - models: Sequence[Runs], - config: dict[str, Any] | None = None, + agents: Mapping[str, Runs], + envs: Mapping[str, str] | None = None, + params: dict[str, Any] | None = None, budget: dict[str, Any] | None = None, ) -> None: ... def forget(self, workspace: str = "") -> bool: ... @@ -52,23 +52,19 @@ def held() -> dict[str, object]: ... def crash(why: BaseException, **said: object) -> None: ... def snag(name: str, **said: object) -> None: ... -# kept.py -- one agent, written down +# kept.py -- one agent, written down: the word `-a` takes after `=` class Runs(NamedTuple): - spec: str - anchor: str = "" - permission: str = "" + spec: str # cli/model:effort provider: str = "" - goals: bool = True - web_search: bool | None = None -def written(runs: Runs) -> dict[str, Any]: ... -def read_back(held: dict[str, Any], *, goals: bool = True) -> Runs | None: ... +def written(runs: Runs) -> str: ... # cli[@provider]/model:effort +def read_back(said: object) -> Runs | None: ... # epic.py -- one run, written down as it happens JOURNAL = "epic.jsonl" # the run's own record, inside the epic RECORD = "epic.{flow}_{ident}.jsonl" # the record of one flow the run called RECORDS = "epic.*.jsonl" +RESUME = "resume.jsonl" # the engine's journal of a resumable run, inside its epic SESSIONS = "sessions" # where each session's own logs are pointed at -STATE = "state.json" TRACES = "traces" LOCAL = "local" # what a session that ran on this machine is anchored as class Session(NamedTuple): @@ -81,15 +77,12 @@ class Session(NamedTuple): flow: str = "" parent: str = "" record: str = "" -class Drove(NamedTuple): +class Drove(NamedTuple): # one agent role, and what it was given agent: str backend: str model: str effort: str - permission: str = "" provider: str = "" - goals: bool = True - person: bool = False @property def spec(self) -> str: ... class Called(NamedTuple): @@ -112,23 +105,27 @@ class Ran(NamedTuple): sessions: tuple[Session, ...] = () called: tuple[Called, ...] = () resumable: bool = False + ref: str = "" # the flow's canonical ref + envs: tuple[str, ...] = () # each `role=spec`, as `-e` spells one + params: dict[str, Any] = {} + budget: dict[str, Any] | None = None + picked_up: str = "" # the epic it was picked up from @property def name(self) -> str: ... -class State(dict[str, Any]): # what a resumable flow writes into, saved as it writes - def __init__( - self, at: Path, flow: str, held: Mapping[str, Any] | None = None - ) -> None: ... - def save(self) -> None: ... class Epic: # a context manager, closed however the run ends def __init__( self, flow: str, - agents: Sequence[AgentBase], task: str, workspace: Path | None = None, *, + ref: str = "", + agents: Sequence[Drove] = (), + envs: Sequence[str] = (), + params: Mapping[str, Any] | None = None, + budget: Mapping[str, Any] | None = None, resumable: bool = False, - picked_up: str = "", + picked_up: Path | None = None, profile: bool = False, ) -> None: ... @property @@ -139,18 +136,16 @@ class Epic: # a context manager, closed however the run ends def record(self) -> str: ... @property def workspace(self) -> Path: ... - def state(self, flow: str = "", held: Mapping[str, Any] | None = None) -> State: ... + @property + def resume(self) -> Path: ... # where the engine keeps this run's journal + def stopped(self) -> None: ... def opened(self, agent: AgentBase, session: str, parent: str = "") -> None: ... + def session( + self, agent: str, backend: str, provider: str, ident: str, parent: str = "" + ) -> None: ... def links(self, only: str = "") -> None: ... def write(self, event: str, **said: Any) -> None: ... - def called( - self, - flow: str, - agents: Sequence[AgentBase], - task: str, - *, - resumable: bool = False, - ) -> Sub: ... + def called(self, flow: str, task: str = "", *, resumable: bool = False) -> Sub: ... def __enter__(self) -> Self: ... def __exit__( self, kind: type[BaseException] | None, why: object, traceback: object @@ -161,12 +156,11 @@ class Sub(Epic): # one flow another flow called, in a record beside that run's under: Epic, record: str, flow: str, - agents: Sequence[AgentBase], - task: str, + task: str = "", *, resumable: bool = False, ) -> None: ... - def ended(self, kind: type[BaseException] | None = None) -> None: ... + def ended(self, kind: type[BaseException] | None = None, how: str = "") -> None: ... def called(agent: str, backend: str, provider: str, ident: str) -> str: ... def under(workspace: Path | str | None = None) -> Path: ... def epics(workspace: Path | str | None = None) -> list[Path]: ... @@ -177,7 +171,8 @@ def sessions(epic: Path) -> list[Session]: ... def opened(epic: Path) -> dict[str, list[str]]: ... def linked(epic: Path) -> dict[str, list[str]]: ... def where(epic: Path, session: Session) -> Path: ... -def state(epic: Path, flow: str = "") -> dict[str, Any]: ... +def picks_up(epic: Path) -> bool: ... +def state(epic: Path, flow: str = "") -> dict[str, Any]: ... # flow by canonical ref def resumed(flow: str, workspace: Path | str | None = None) -> Path | None: ... # exporting.py -- one whole run as one archive @@ -196,33 +191,53 @@ def logged(epic: Path) -> dict[str, dict[str, Path]]: ... def plain(said: str, struck: Sequence[str] = ()) -> str: ... def sized(count: int) -> str: ... -# runner.py -- a flow loaded and handed its agents +# runner.py -- a flow loaded, handed a driver per role, and run under an epic +class Refused(ValueError): ... # a run refused before anything of it ran +class Line(NamedTuple): # an `hmz exec` line, read + flow: str + task: str + agents: tuple[AgentSpec, ...] = () + envs: tuple[EnvSpec, ...] = () + params: dict[str, str] = {} + budget: Budget | None = None + resume: bool = False + as_json: bool = False +def read_line(argv: list[str]) -> Line: ... class Runner: def __init__( self, flow: str | os.PathLike[str], - agents: Sequence[AgentBase], - config: BaseModel | dict[str, Any] | None = None, - resume: str | os.PathLike[str] | None = None, - container: str = "", - budget: Allowance | Mapping[str, Any] | None = None, + *, + agents: Mapping[str, str | AgentDriver] | Iterable[AgentSpec] = (), + envs: Mapping[str, str | EnvDriver] | Iterable[EnvSpec] = (), + params: Mapping[str, Any] | FlowParams | None = None, + budget: Budget | Mapping[str, Any] | None = None, + resume: bool | str | os.PathLike[str] = False, + workspace: str | os.PathLike[str] | None = None, ) -> None: ... - @property - def agents(self) -> tuple[AgentBase, ...]: ... - @property - def budget(self) -> Allowance: ... - @property - def unwatched(self) -> bool: ... + flow: str; impl: FlowImpl; declaration: Declaration; agents: dict[str, AgentDriver] + envs: dict[str, EnvDriver]; params: FlowParams; budget: Budget + picked_up: Path | None; workspace: Path; recorder: Recorder | None # properties def unreadable(self) -> str: ... - def unserved(self) -> str: ... - def run(self, task: str) -> None: ... -def read_agent(spec: str) -> tuple[str, Profile, str, str, str]: ... -def flow_and_agents( - argv: list[str], -) -> tuple[str, list[AgentBase], str, dict[str, Any] | None, Allowance | None, bool]: ... -def set_up_from( - said: str | os.PathLike[str], -) -> tuple[dict[str, Any] | None, Allowance | None]: ... + def watch(self, listener: Listener) -> None: ... + async def arun( + self, + task: str, + *, + outworlder: OutworlderDriver | None = None, + opened: Callable[[str, AgentBase, SessionBase], None] | None = None, + started: Callable[[Epic], None] | None = None, + ) -> Any: ... + def run(self, task: str, *, outworlder: OutworlderDriver | None = None) -> Any: ... +class Recorder: # answers to runtime/flowing's Recorder, writing the epic + started: bool + def entered(self, call: LiveCall) -> None: ... + def left(self, call: LiveCall, error: BaseException | None) -> None: ... + def spawned(self, call: LiveCall, role: str, session: SessionHandle, + driver: AgentDriver) -> None: ... + @property + def sessions(self) -> tuple[SessionHandle, ...]: ... + def usage(self) -> Usage: ... ``` ## Requirements @@ -231,14 +246,15 @@ def set_up_from( MUST drive no coding agent itself: every turn MUST be taken through `coganchor`. - MUST NOT name `cli`, `daemon`, `tui` or `sdk`, MUST restate no rule the layers under it carry out, and MUST load nothing until it is named. `telemetry` MUST name nothing above it. -- `Settings` MUST answer what a workspace was last set up to run — the flow, each of its - agents and where its work lands, the flow's setup, what a run may spend — and MUST answer +- `Settings` MUST answer what a workspace was last set up to run — the flow, what each of its + agent and environment roles was given, its params, what a run may spend — and MUST answer what is not a workspace's at all. - A setting that is a question somebody has to answer MUST have three answers — yes, no, and nobody asked — and reading one MUST NOT write it. - Two holders of the settings MUST NOT put back what the other has written; remembering a - flow's agents MUST leave its config and budget alone where neither is handed in, an empty - one MUST erase, and settings humanize did not write MUST read as nothing remembered. + flow's agents MUST leave its environments, params and budget alone where they are not handed + in, an empty one MUST erase, and settings humanize did not write MUST read as nothing + remembered. - Nothing MUST be reported where the question has not been answered yes, a run with nobody at a terminal MUST NOT ask, and `SAYS` MUST answer it for one process alone. - `SENT` and `KEPT` MUST be the whole of what is sent and what never is, in words, readable @@ -249,9 +265,9 @@ def set_up_from( made; what is not a failure MUST be reportable too, as counts and names. A reporter that will not start, a callable that raises and a report that cannot be sent MUST each leave the run as it was. -- An agent written down MUST be a CLI, an account, a model at an effort and the machine its - work lands on, and nothing else. A field that says nothing MUST read back as that field's - own silence, and an entry older than a setting MUST read as what every agent did then. +- An agent written down MUST be a CLI, an account and a model at an effort -- the word `-a` + takes after its role -- and nothing else; what it may do is its role's, and where it works + is its environment's. An entry that is not one MUST read back as nothing. - One epic MUST be one run: opened when the flow starts, closed however the run stops, never reopened. Epics MUST read back in the order they were run, and anything else under them MUST read as no run rather than fail. @@ -260,8 +276,9 @@ def set_up_from( - Every session a run opened MUST read back as whose it was, what took its turns, which account they ran as, what the backend called it and which conversation it was forked from, and its own logs MUST be reachable from the epic without humanize copying or moving them. -- What a resumable flow leaves MUST be kept under that flow's own name, MUST be there for the - next run of it even where this one was killed, and MUST NOT be able to stop a run. +- A resumable run's journal MUST be kept inside its epic, MUST be there for a run picking it + up even where this one was killed, and a run picking one up MUST be handed a copy of it in an + epic of its own and MUST say which epic it came from. - A bundle MUST be one archive readable on a machine that was not there: everything the run wrote, the session logs themselves, and a manifest saying what the run was down to each backend's version and the hash of the executable that took its turns. @@ -272,25 +289,23 @@ def set_up_from( - A bundle MUST be written whole and leave nothing behind where it fails, MUST be readable by whoever exported it alone, MUST land where somebody is standing unless a path was named, and MUST replace the earlier archive when one run is exported twice. -- Constructing a `Runner` MUST raise `NotAFlow`, before anything runs, for a file that is not - a flow, a number of agents the flow does not drive, an agent that cannot run a moment or a - goal the flow declared, a config the flow did not ask for, or a skill that cannot be - fetched. Whatever the flow itself raises as it loads MUST be left alone. -- What the flow declared about each place MUST be settled onto its agent before the first turn - and over whatever it was made with; what a lenient place gave up MUST be answerable in - words, an agent the flow names MUST answer to that name from then on, and the person a flow - talks to MUST be made here rather than given and MUST be among `agents`. -- `run` MUST drive the flow with the agents as it declared them until it returns, one written - as a coroutine among them, MUST write the run down as it happens, and the run MUST be over - when it returns. A run given a container MUST work in one throughout and take it down - however it ends; one given none MUST start none. -- Every run MUST be held to one allowance across every agent of it; what the flow declared - MUST be a default whoever started the run may override, and a run nothing will stop MUST be - answerable as such before it starts. `set_up_from` MUST lift that allowance out of a `-c` - file before the flow's own model sees it. -- A resumable flow MUST be handed what the run it picks up left behind — the last run of it in - this workspace unless one was named — and what it writes MUST belong to the run writing it. -- `flow_and_agents` MUST read the whole `hmz exec` line, MUST NOT load a flow to answer - `--help`, MUST split an `-a` naming several before reading any, MUST order agents that named - a place as the flow does and refuse a name it has not got, and MUST hand the run's allowance - back beside the flow's setup rather than folded into it. +- Constructing a `Runner` MUST raise `Refused`, before anything runs and before any agent + starts, for a flow that cannot be loaded; a role given that the flow does not declare, that + the runtime fills -- an `Outworlder`, a `LocalEnv` -- or that is given twice; a required + role left out; an agent that is not the harness its role names or whose harness does not + serve what its role asks; a spec no driver can be made for; params the flow does not take; + no budget, except for a flow humanize ships, which runs under `Budget(cost=inf)`; and a + run to pick up that is not there, or of a flow that is not resumable. What the flow itself + raises as it is imported MUST be refused with its reason. `arun` MUST raise `Refused` too + for an environment that cannot be reached and for anything the engine refuses before the + flow is called. +- `arun` MUST probe every environment it was given before the flow is called, MUST run the + flow over the drivers with the workspace as every `LocalEnv` role and whoever is outside + the run as every `Outworlder` role -- nobody, away, where none was given -- MUST write the + run down as it goes: each flow call a record under the one that made it, each session in + the record of the call that opened it and named for its role, and what the run spent; and + MUST close every driver it was given however the run ends. A run stopped from outside, or + by its budget, MUST be written down as stopped rather than failed. +- `read_line` MUST read the whole `hmz exec` line, MUST NOT load a flow to answer `--help`, + MUST take every `-a`, `-e`, `-p` and `-b` as one list however they were broken up, and + MUST refuse one that cannot be read as argparse refuses a line. diff --git a/specs/runtime/doing.md b/specs/runtime/doing.md index 5bfb6f16..c590502c 100644 --- a/specs/runtime/doing.md +++ b/specs/runtime/doing.md @@ -30,46 +30,45 @@ class Hmz: def epics(self) -> Epics: ... def backends(self) -> tuple[Profile, ...]: ... def reports(self) -> bool: ... - def read( - self, argv: list[str] - ) -> tuple[ - str, list[AgentBase], str, dict[str, Any] | None, Allowance | None, bool - ]: ... + def read(self, argv: list[str]) -> Line: ... def runner( self, flow: str | os.PathLike[str], - agents: Sequence[AgentBase], - config: BaseModel | dict[str, Any] | None = None, - resume: str | os.PathLike[str] | None = None, - container: str = "", - budget: Allowance | Mapping[str, Any] | None = None, + *, + agents: Mapping[str, str | AgentDriver] | Iterable[AgentSpec] = (), + envs: Mapping[str, str | EnvDriver] | Iterable[EnvSpec] = (), + params: Mapping[str, Any] | FlowParams | None = None, + budget: Budget | Mapping[str, Any] | None = None, + resume: bool | str | os.PathLike[str] = False, ) -> Runner: ... def run( self, flow: str | os.PathLike[str], - agents: Sequence[AgentBase], task: str, - config: BaseModel | dict[str, Any] | None = None, - resume: str | os.PathLike[str] | None = None, - container: str = "", - budget: Allowance | Mapping[str, Any] | None = None, + *, + agents: Mapping[str, str | AgentDriver] | Iterable[AgentSpec] = (), + envs: Mapping[str, str | EnvDriver] | Iterable[EnvSpec] = (), + params: Mapping[str, Any] | FlowParams | None = None, + budget: Budget | Mapping[str, Any] | None = None, + resume: bool | str | os.PathLike[str] = False, + outworlder: OutworlderDriver | None = None, ) -> Run: ... - def exec(self, argv: list[str]) -> None: ... + def exec(self, argv: list[str]) -> Any: ... # running.py -- one run of one flow class Run: - def __init__(self, runner: Runner, task: str) -> None: ... - @property - def agents(self) -> tuple[AgentBase, ...]: ... - @property - def unwatched(self) -> bool: ... - @property - def running(self) -> bool: ... + def __init__( + self, runner: Runner, task: str, *, outworlder: OutworlderDriver | None = None + ) -> None: ... + flow: str; ref: str; task: str; declaration: Declaration; budget: Budget + usage: Usage; epic: Path | None; running: bool; raised: BaseException | None + result: Any # properties @property - def raised(self) -> BaseException | None: ... + def agents(self) -> tuple[AgentBase, ...]: ... # behind each session opened, by role def unreadable(self) -> str: ... - def unserved(self) -> str: ... - def run(self) -> None: ... + def watch(self, listener: Listener) -> None: ... + def opened(self, callback: Callable[[str, AgentBase, SessionBase], None]) -> None: ... + def run(self) -> Any: ... def start(self) -> None: ... def wait(self, timeout: float | None = None) -> bool: ... def stop(self) -> None: ... @@ -85,6 +84,7 @@ class Epics: def opened(self, epic: Path) -> dict[str, list[str]]: ... def state(self, epic: Path, flow: str = "") -> dict[str, Any]: ... def resumed(self, flow: str) -> Path | None: ... + def picks_up(self, epic: Path) -> bool: ... def traced( self, epic: Path, @@ -120,20 +120,10 @@ class Flows: def all(self) -> list[Offer]: ... def find(self, named: str) -> str: ... def about(self, named: str) -> str: ... - def places(self, named: str | os.PathLike[str]) -> tuple[Place, ...]: ... - def check( - self, named: str | os.PathLike[str], *, static: bool = False - ) -> tuple[Finding, ...]: ... - def prophecy(self, named: str | os.PathLike[str]) -> Prophecy | None: ... - def foretell(self, named: str | os.PathLike[str]) -> str: ... - def configures(self, named: str | os.PathLike[str]) -> type[BaseModel] | None: ... - def declared(self, named: str | os.PathLike[str]) -> Allowance | None: ... + def declared(self, named: str | os.PathLike[str]) -> Declaration: ... def resumes(self, named: str | os.PathLike[str]) -> bool: ... def fork(self, named: str, into: str | os.PathLike[str] | None = None) -> str: ... - def running(self) -> tuple[Running, ...]: ... - def set_up_from( - self, said: str | os.PathLike[str] - ) -> tuple[dict[str, Any] | None, Allowance | None]: ... + def running(self) -> tuple[LiveCall, ...]: ... class Flowverses: def all(self) -> list[Flowverse]: ... def nearest(self) -> list[Flowverse]: ... @@ -218,14 +208,17 @@ class Fallbacks: nobody named MUST follow a flow that changes directory. Running an `hmz exec` line and building a run out of its parts MUST leave the caller holding the same run. - Making a `Run` MUST start nothing: `run()` MUST drive the flow to its return where it was - called, `start()` MUST drive it on a thread of its own and refuse a second start. -- What a run started on a thread raised MUST be kept and answerable rather than swallowed, - and a run in a container MUST be one at a time per process. -- `stop()` MUST leave the turn now running to close out and the loop to end; `close()` MUST - NOT wait for it and MUST end every conversation still open. -- `check` MUST be both readings of a flow in one answer, the second left out where the first - found an error and nothing said twice; `prophecy` MUST answer with nothing for a flow that - is not an atlas or does not compile, and `foretell` MUST refuse to write where there is none. + called -- on a loop of its own, or on a thread of its own where the caller's thread is + already running one -- and `start()` MUST drive it on a thread of its own and refuse a + second start. +- What a run started on a thread raised, or returned, MUST be kept and answerable rather than + swallowed. Each session the run opens MUST be told to whoever asked with `opened` before + its first turn, and every event of every session to whoever asked with `watch`. +- `stop()` MUST cancel the flow from any thread -- the turn under way interrupted, the flow + unwinding and what it made let go of in its own time; `close()` MUST NOT wait for it and + MUST end every conversation still open. +- `declared` MUST be what the flow declares, the roles the runtime fills marked, read off the + flow as it is now; a flow that cannot be loaded MUST raise what the flow API says of it. - Where a flowverse came from MUST be answered here, with whatever was signed into a URL taken out of it, and asking a backend what it runs as one account MUST be reached through this. - A trace of one run MUST be gathered here rather than by whoever asked for it, by the ids the diff --git a/specs/runtime/flowing.md b/specs/runtime/flowing.md index 440c0887..f5186055 100644 --- a/specs/runtime/flowing.md +++ b/specs/runtime/flowing.md @@ -6,10 +6,6 @@ writes against is `hmz.flows` (see [flows.md](../flows.md)), whose `flow`, `load `Outworlder.new` hand their calls to the engine here. Nothing here drives a coding agent itself: the harness drivers are written against `hmz.coganchor`. -The previous flow API's machinery -- `checking`, `driving`, `prophecy`, `prophesying`, -`stepping`, `proving` -- is still here, written against `hmz._legacy_flows`, and goes when -every way in has moved over; its part of this spec is marked *legacy* below. - ## API ```python @@ -236,9 +232,8 @@ def standing(at: Path) -> str: ... def plain(url: str) -> str: ... # a URL with whatever was signed into it taken out # finding.py -- which flow a name means -BUILTIN_AT: Path # where humanize's own flows are +BUILTIN_AT: Path # where humanize's own flows are: hmz/flows/builtin ENTRY = "__init__.py" # what a flow's own directory is entered by -PROPHECY = "prophecy.pkl" # legacy: what a shipped prophecy is called beside it class Offer(NamedTuple): whose: str # the flowverse it comes from name: str @@ -249,13 +244,11 @@ def offered(under: Path) -> list[str]: ... def entry(under: Path, name: str) -> Path | None: ... def within(one: Flowverse, name: str) -> Path | None: ... def find(named_: str) -> str: ... # what runs -def reading(named_: str) -> str: ... # what a reading is pointed at -def foretold(named_: str) -> str: ... # legacy: what a prophecy is compiled out of def at(named_: str) -> str: ... # its own directory, or "" -def inside(named_: str) -> str: ... # which of the flows a file holds +def inside(named_: str) -> str: ... # which of the flows a module holds def about(named_: str) -> str: ... -def held(where_: str | os.PathLike[str]) -> list[Flow]: ... # legacy mark -def loaded(where_: str | os.PathLike[str]) -> dict[str, Any]: ... # legacy +def resolved(named_: str) -> FlowImpl: ... # the flow a way in runs, loaded +def builtin(flow: FlowImpl) -> bool: ... # whether humanize ships it def fork(named_: str, into: str | os.PathLike[str] | None = None) -> str: ... # skills.py -- what a flow brings its agents; `CARD`, `SKILLS` and `Loaded` come from @@ -266,16 +259,6 @@ def fetched(url: str) -> Path: ... def under() -> Path: ... ``` -*Legacy* -- written against `hmz._legacy_flows`, removed with it: `checking.py` (`Finding`, -`Capability`, `checked`, `catalogue`, `briefed`, `surface`, `offered`, `misplaced`), -`driving.py` (`Entry`, `NotAFlow`, `Running`, `Place`, `drives`, `wanted`, `configures`, -`resumes`, `declared`, `load`, `running`, `container`, `declares`, `readies`, `carries`, -`set_up`, `serves`, `lands`, `runs_at`, `comes_to`, `contained`, `entered`, `left`, -`lands_in`), `prophecy.py` (`Field`, `Shape`, `Reads`, `When`, `Node`, `Edge`, `Prophecy`, -`Shipped`, `canonical`, `digest`, `kept`, `told`, `shipped`), `prophesying.py` (`Prophesied`, -`prophesied`, `is_atlas`, `named_as`), `stepping.py` (`walking`) and `proving.py` (`Scenario`, -`Outcome`, `Proof`, `proved`), with the signatures they have today. - ## Requirements ### Defining a flow @@ -309,7 +292,8 @@ def under() -> Path: ... machine short of a declared resource (`ResourceUnmet`), and another harness for a role typed as one (`HarnessMismatch`). A refusal for one kind of agent MUST be worked out once. - An `Outworlder` role left out MUST be the run's own outworlder, and one given `Outworlder.new()` - that one. A `LocalEnv` role left out MUST be the run's workspace. + that one. A `LocalEnv` role left out MUST be the run's workspace, and one given an + environment on another machine MUST be refused with `CapabilityMissing`. - Params MUST be taken as they are when they are the callee's own class, and validated otherwise -- from another model, or a mapping whose string values are read as the fields' types or as JSON -- raising `ParamsError` when they do not validate. @@ -400,9 +384,15 @@ def under() -> Path: ... ### Where flows come from - Every flow MUST be listable under one name apiece -- humanize's own by a bare name, every - other as `/`, a file holding several as `:` -- and a name - MUST resolve nearest first: this project's flows, then yours, then the rest. A name - qualified by a flowverse MUST NOT be stood in for, and a path MUST be taken outright. + other as `/`; the flow a bare name means under its module's own name and + every other visible flow of the module as `:`, a hidden one not at all -- and + a name MUST resolve nearest first: this project's flows, then yours, then the rest. A name + qualified by a flowverse MUST NOT be stood in for, and a path MUST be taken outright. A + module that will not import MUST still be listed, under its name. +- `resolved` MUST load what a way in names -- a name, `/`, either with + `:`, a path, or a VCS ref, fetched on the calling thread -- MUST say so where a + name nothing answers to could have come from a flowverse not fetched yet, and MUST mark the + flows humanize ships with `full_view`; nothing else is. - `brought` MUST bring the flow's own skills first and the ones it named after, the flow's own winning a shared name, and MUST raise where one cannot be fetched rather than at the turn. - `fork` MUST copy the whole of a flow and MUST refuse a name already taken rather than write @@ -411,22 +401,3 @@ def under() -> Path: ... - Every name above MUST be fetched only when it is named, so that listing flows costs nothing that reading or driving one does. Nothing here MAY be imported by `hmz.flows` at import, or drive an agent. - -### Legacy - -- *(legacy)* A flow MUST be read as it is when it is asked for, never cached, and a file that - will not read MUST be one line of a list rather than the end of it. -- *(legacy)* `checked` MUST import and execute nothing of the flow it reads, MUST answer with - findings rather than raise, and MUST keep `error` for a flow that cannot run, cannot be - answered or cannot end. A flow with no error finding MUST be one `load` would take. -- *(legacy)* What a flow declares MUST be readable before an agent is chosen, and MUST raise - `NotAFlow` otherwise. An agent that cannot serve a place MUST be refused, saying why. -- *(legacy)* `catalogue` MUST report what this installation serves at the moment it is asked. -- *(legacy)* `driving.load` MUST refuse a name nothing answers to where the flow is asked for, - run the called flow as a flow of its own, and give the caller's agents back as they were. -- *(legacy)* `driving.running` MUST answer with the branch it is asked from inside a flow and - with every flow of the run from outside; `container` MUST answer with the workspace as the - machine a contained run works on has it. -- *(legacy)* A prophecy MUST be canonical, a shipped one MUST be what runs, and a run MUST be - picked up only into the prophecy it was doing. `proved` MUST drive the flow in a process of - its own, one per scenario, and MUST end every proof. diff --git a/specs/sdk.md b/specs/sdk.md index eb9016e6..80708f7e 100644 --- a/specs/sdk.md +++ b/specs/sdk.md @@ -9,8 +9,8 @@ what it offers -- every answer is `hmz.runtime`'s or `hmz.daemon`'s. ```python # __init__.py -- both ways in, and every type either hands back __all__ = ["Accounts", "Daemon", "Daemons", "Epics", "Fallbacks", "Flows", "Flowverses", - "Held", "Hmz", "Run", "Session"] -def __getattr__(name: str) -> object: ... + "Held", "Hmz", "Refused", "Run", "Session", "fakes"] +def __getattr__(name: str) -> object: ... # `fakes` is `hmz.runtime.flowing.fakes`, whole # daemons.py -- the runs being held apart from a terminal class Daemons: diff --git a/specs/tui.md b/specs/tui.md index 90db451c..a4ad005c 100644 --- a/specs/tui.md +++ b/specs/tui.md @@ -12,15 +12,17 @@ class Humanize(App[None]): def __init__( self, flow: str = "", - agents: Sequence[Runs] = (), - config: BaseModel | None = None, + agents: Mapping[str, Runs] | None = None, # by role + params: BaseModel | None = None, # the flow's params session: Session | None = None, ) -> None: ... def reattached(self) -> None: ... def action_quit(self) -> None: ... + def said(self) -> dict[str, Any]: ... # the flow, its budget and usage, as JSON ``` -Textual's `run()` opens it; the other two are what a run held elsewhere calls to redraw or stop. +Textual's `run()` opens it; the other three are what a run held elsewhere calls to redraw, stop, +or say what it is running. ## Requirements @@ -52,20 +54,27 @@ Textual's `run()` opens it; the other two are what a run held elsewhere calls to - MUST offer exactly these commands, each doing what it says: `/flow`, `/btw`, `/flowverses`, `/providers`, `/fallback`, `/epics`, `/resume`, `/settings`, `/monitor`, `/clear`, `/details`, `/afk`, `/stop`, `/exit`. `/btw` MUST be answered from a snapshot, not by asking the flow. -- MUST carry the last run of this directory on for `/resume` — its flow, agents, task and what it - left behind, saying which — say why there is none to carry on, and refuse it, as it refuses - picking any run up, while a flow is running or stopping. +- MUST carry the last run of this directory of a flow that can be picked up on for `/resume` — + its flow, roles, params, budget and task, picking up its journal, saying which — say why there + is none to carry on, and refuse it, as it refuses picking any run up, while a flow is running + or stopping. - MUST let a flow's agents be set up whatever is happening, but offer a flow choice only when idle. - MUST read the flows a place at a time — every flowverse fetched or not, then this project's own — showing which place is read, and letting a flow be copied here whole and under its name. -- MUST apply a saved menu to running agents from their next turn on, say that a changed CLI takes - effect only from the next run, and refuse to save a flow whose agent names no model. -- MUST let what a run may spend be set and read on the page its agents are on, asking once before - saving a run nothing will stop unless the flow says it is meant to run unbounded. +- MUST set a flow up by its roles: one row per agent role and one per environment role the flow + declares, leaving out the ones the runtime fills -- an `Outworlder`, a `LocalEnv` -- then its + params, asked with the flow's own params model, and what a run may spend. A saved menu MUST take + effect from the next run, and MUST refuse to save a flow whose agent role names no model, or -- + for every flow but `chat` -- one that has no budget. +- MUST let what a run may spend -- a duration, a cost, output tokens, and whether a turn is let + finish -- be set and read on the page its roles are on. - MUST make an agent a CLI, an account, a model and an effort and nothing else, offering only - CLIs installed here, this machine's own account as `as local`, models known runnable as the - chosen account, and efforts that model takes — returning unchanged whatever the flow said and - the sheet never asked, and letting an account be made where one is asked for. + CLIs installed here whose harness is the one the role names and serves what the role asks, + this machine's own account as `as local`, models known runnable as the chosen account, and + efforts that model takes -- and letting an account be made where one is asked for. An + environment role MUST take a spec as `-e` spells one. +- MUST be whoever is outside a run: a question a flow puts to its outworlder MUST be asked at + the prompt and answered with the next line typed, and `/afk` MUST make the outworlder away. - MUST offer, per place flows come from, what it holds, adding one, fetching it again and taking one away, against the same store the flows are read from, with any credential in a URL hidden. - MUST list every account under its CLI and offer correcting, re-signing, what it falls back to diff --git a/src/hmz/_legacy_flows/__init__.py b/src/hmz/_legacy_flows/__init__.py deleted file mode 100644 index d2c95d63..00000000 --- a/src/hmz/_legacy_flows/__init__.py +++ /dev/null @@ -1,405 +0,0 @@ -"""The whole of what a flow imports: what it drives, the mark, and the words for a turn. - -A flow is a directory: an `__init__.py` that is the flow itself, whatever that imports beside -it, and a `skills/` of the skills it brings -- laid out the way every one of these CLIs lays a -skill out, one directory apiece with a `SKILL.md` in it. So a flow is a thing that can be -copied, forked and edited whole, and what it needs to do its work travels with it. - -A flow is a function marked with :func:`flow`, and nothing else is one. `@flow()` is the flow -its directory holds under the directory's own name; `@flow(name="draft")` is one of several it -holds, called `:draft` -- so that three phases of one thing live in one flow and are -three things to run. What the function is called is the flow's own business: `run`, `main`, -`draft_it`, all the same to a name that never mentions it. - -And this is the whole of what a flow imports:: - - from hmz._legacy_flows import Agent, Moment, flow - - @flow - def run(agents: tuple[Agent, Agent], task: str) -> None: - ... - -One import rather than four, because a flow is written against one thing: what it drives, what -it may ask of it, and what it is worth saying about a turn. Which of humanize's own modules any -of that is written in is humanize's business -- a flow that named them would be a flow that -breaks when one of them moves, and a flow is somebody else's repository. - -So what is here is only ever what writing a flow takes, and it is a short list. The interfaces -a flow drives, in [agent.py](agent.py). The marks an atlas declares its graph with, in -[atlas.py](atlas.py). The mark that makes a function a flow, and what that mark says, here. -Everything else is handed through: the vocabulary a turn is described in from -:mod:`hmz.coganchor.agents`, the facts about the CLIs and what each of them runs, where -humanize keeps what outlives a run, and calling another flow -- which is `load`, and is -:mod:`hmz.runtime.flowing.driving`'s. - -What is *not* here is everything humanize does *to* a flow. Finding one, listing them, reading -one without running it, driving one, compiling an atlas, fetching the skills a flow named, -walking a prophecy: none of it is a thing a flow names, so none of it is a thing a flow can -break on. All of it is :mod:`hmz.runtime.flowing`, which is written against this and which this -never imports at the top of the file. - -What is handed through is fetched when a flow names it rather than imported with this module. -This is also what a command line is routed through before it knows whether it names a flow at -all, so importing it must cost no more than reading a directory. -""" - -from __future__ import annotations - -from dataclasses import dataclass -from typing import TYPE_CHECKING, overload - -from .agent import Agent, Driven, Person, Session -from .atlas import Atlas, Kind, Marked, Sub, atlas, logic, mind, sub - -if TYPE_CHECKING: - from collections.abc import Callable, Iterable - - from hmz import home - from hmz.coganchor import backends, models - from hmz.coganchor.agents import ( - EVERYWHERE, - PERMISSIONS, - SWARM, - UNSAID, - WINDOW, - AgentConfig, - AgentDefaults, - Allowance, - Board, - Budget, - Event, - Failed, - Goal, - Hook, - Hooks, - HumanAgent, - Hung, - Isolated, - Item, - Moment, - Needs, - Occasion, - Question, - Refused, - Remote, - Stopped, - Tool, - Unhooked, - Unrecoverable, - Usage, - Verdict, - ) - from hmz.coganchor.backends import Model, Profile - from hmz.runtime.flowing.driving import NotAFlow, Running, container, load, running - -__all__ = [ - "EVERYWHERE", - "PERMISSIONS", - "SWARM", - "UNSAID", - "WINDOW", - "Agent", - "AgentConfig", - "AgentDefaults", - "Allowance", - "Atlas", - "Board", - "Budget", - "Driven", - "Event", - "Failed", - "Flow", - "Goal", - "Hook", - "Hooks", - "HumanAgent", - "Hung", - "Isolated", - "Item", - "Kind", - "Marked", - "Model", - "Moment", - "Needs", - "NotAFlow", - "Occasion", - "Person", - "Profile", - "Question", - "Refused", - "Remote", - "Running", - "Session", - "Stopped", - "Sub", - "Tool", - "Unhooked", - "Unrecoverable", - "Usage", - "Verdict", - "atlas", - "backends", - "container", - "flow", - "home", - "load", - "logic", - "mind", - "models", - "running", - "sub", -] - -#: The two modules of humanize's own that a flow reaches through here whole: what each CLI -#: is, and what each of them runs. A loop that turns the effort down when a model starts -#: writing less asks the second of them what rungs there are, which is a question about a -#: backend rather than about any agent -- so it is handed through as it stands rather than -#: flattened into a name apiece. Under the name the flow writes rather than the one the -#: module is at: a flow says `flows.backends`, and where humanize keeps that is humanize's -#: own to move. -_MODULES = {"backends": "hmz.coganchor.backends", "models": "hmz.coganchor.models"} - -#: And the names a flow imports from here that are written down elsewhere: the vocabulary a -#: turn is described in, where humanize keeps what outlives a run, and what it takes for one -#: flow to run another -- which is the runtime's, being the thing that drives a flow, and is -#: handed through so that a flow calling a flow writes the one import it already has. -_ELSEWHERE = { - "AgentConfig": "hmz.coganchor.agents", - "AgentDefaults": "hmz.coganchor.agents", - "Allowance": "hmz.coganchor.agents", - "Board": "hmz.coganchor.agents", - "Budget": "hmz.coganchor.agents", - "EVERYWHERE": "hmz.coganchor.agents", - "Event": "hmz.coganchor.agents", - "Failed": "hmz.coganchor.agents", - "Goal": "hmz.coganchor.agents", - "Hook": "hmz.coganchor.agents", - "Hooks": "hmz.coganchor.agents", - "HumanAgent": "hmz.coganchor.agents", - "Hung": "hmz.coganchor.agents", - "Isolated": "hmz.coganchor.agents", - "Item": "hmz.coganchor.agents", - "Model": "hmz.coganchor.backends", - "Moment": "hmz.coganchor.agents", - "Needs": "hmz.coganchor.agents", - "NotAFlow": "hmz.runtime.flowing.driving", - "Occasion": "hmz.coganchor.agents", - "PERMISSIONS": "hmz.coganchor.agents", - "Profile": "hmz.coganchor.backends", - "Question": "hmz.coganchor.agents", - "Refused": "hmz.coganchor.agents", - "Remote": "hmz.coganchor.agents", - "Running": "hmz.runtime.flowing.driving", - "SWARM": "hmz.coganchor.agents", - "Stopped": "hmz.coganchor.agents", - "Tool": "hmz.coganchor.agents", - "UNSAID": "hmz.coganchor.agents", - "Unhooked": "hmz.coganchor.agents", - "Unrecoverable": "hmz.coganchor.agents", - "Usage": "hmz.coganchor.agents", - "Verdict": "hmz.coganchor.agents", - "WINDOW": "hmz.coganchor.agents", - "container": "hmz.runtime.flowing.driving", - "home": "hmz", - "load": "hmz.runtime.flowing.driving", - "running": "hmz.runtime.flowing.driving", -} - - -def __getattr__(name: str) -> object: - """Hands through what a flow imports from here that is written down elsewhere. - - Fetched when it is asked for rather than imported at the top of this file, because this - module is also what a list of flows is drawn from and what `hmz exec --help` loads to say - what the line takes: importing it must not cost every coding agent driver there is. A flow - that actually names one of these is a flow about to be run, and pays for it then. - - Args: - name: What was asked for. - - Returns: - The same object the module it is written in holds, so that a flow and humanize itself - are talking about one thing -- `Moment.STOP` here is `Moment.STOP` there. - - Raises: - AttributeError: If nothing here is called that, as for any other module. - """ - from importlib import import_module - - if (whole := _MODULES.get(name)) is not None: - return import_module(whole) - where_ = _ELSEWHERE.get(name) - if where_ is None: - raise AttributeError(f"module {__name__!r} has no attribute {name!r}") - return getattr(import_module(where_), name) - - -@dataclass(frozen=True, slots=True) -class Flow: - """What a flow says about itself where it is written. - - Attributes: - name: What it is called inside its own directory, which is the half after the colon. - "" for the one it holds under the directory's own name, which is what `@flow()` marks. - about: One line saying what it does, for whoever is choosing between them. Read off the - function's own docstring where the decorator was not told one, and off the module's - where the flow is one flow and its function says nothing. - skills: The skills it works by that live somewhere else, each a git repository anything - can clone with an optional `#` saying which of the ones in it is wanted. What - the flow keeps in its own `skills/` is not among them: that is every flow in the - directory's, and is found by looking rather than by being declared. - resumable: Whether it can be picked up where the last run of it left off. One that says - so is handed a dict as its last argument -- what it wrote there last time -- which is - kept in the run's own epic and read back into the run after it. A flow that says - nothing is run from the top every time, which is what every flow was before this. - selectable: Whether people are offered this flow in lists and the flow picker. An - internal composition may set this false while remaining callable by name. - budget: What the flow says a run of it may spend, or None for a flow with no opinion -- - which is every flow written before there was such a thing, and which runs under - whatever this workspace was set up with. Three states rather than two, and the third - is the whole of the exemption from being asked about an unbounded run: an - `Allowance()` written out is a flow saying in its own file that it is *meant* to run - under nothing, which is what `chat` is. A flow never holds itself to it -- the run - does, whatever the flow said -- so this is a default and not an implementation. - """ - - name: str = "" - about: str = "" - skills: tuple[str, ...] = () - resumable: bool = False - selectable: bool = True - budget: Allowance | None = None - - -#: Where a decorated function keeps what it said about itself. On the function rather than in -#: a table, because a file is read by running it: a table would be one more thing to find, -#: and this travels with the thing it describes. -_SAID = "__humanize_flow__" - - -@overload -def flow[**P, T](call: Callable[P, T], /) -> Callable[P, T]: ... - - -@overload -def flow[**P, T]( - *, - name: str = "", - about: str = "", - skills: Iterable[str] = (), - resumable: bool = False, - selectable: bool = True, - budget: Allowance | None = None, -) -> Callable[[Callable[P, T]], Callable[P, T]]: ... - - -def flow[**P, T]( - call: Callable[P, T] | None = None, - /, - *, - name: str = "", - about: str = "", - skills: Iterable[str] = (), - resumable: bool = False, - selectable: bool = True, - budget: Allowance | None = None, -) -> Callable[P, T] | Callable[[Callable[P, T]], Callable[P, T]]: - """Marks a function as a flow. Nothing else is one. - - Written with no name, it is the flow its file holds under the file's own name:: - - @flow - def run(agents: tuple[Agent], task: str) -> None: - ... - - is `ralph_loop`, in `ralph_loop/__init__.py`. Written with one, it is one of several that - flow holds, and is called `:`:: - - @flow(name="gen-idea", about="opens a loose idea into a repo-grounded draft") - def first_pass(agents: Agents, task: str) -> None: - ... - - is `humanize1:gen-idea`. What the function is called is the flow's own business either - way: a name that is written down where a flow is run is a name to keep, and one taken - from the function would change under whoever renamed it. - - A flow may also name skills that live somewhere else, which are mounted onto every session - its agents open alongside the ones in its own `skills/`:: - - @flow(skills=("https://github.com/humanfia/flowverse#deep-research",)) - - And a flow may say that it can be picked up where the last run of it left off, which is - what a loop that is meant to run for a week is:: - - @flow(resumable=True) - def run(agents: tuple[Agent], task: str, state: dict[str, Any]) -> None: - state["round"] = state.get("round", 0) + 1 - - Such a flow is handed a dict as its last argument -- after the config, for one that takes - a config -- holding whatever it wrote there last time. It is kept in the run's own epic - and saved as the flow writes it, so a run that was stopped or killed is one the next run - picks up from rather than one whose week is gone. - - A helper used only by another flow remains callable without cluttering the flow picker:: - - @flow(name="engine", selectable=False) - def engine(agents: tuple[Agent], task: str) -> None: - ... - - And a flow may say what a run of it is worth, which whoever runs it can then override:: - - @flow(budget=Allowance(hours=6, tokens=10.0, dollars=50)) - - Saying nothing is a flow with no opinion, and it runs under whatever the workspace was - set up with. Saying `Allowance()` is a flow claiming it is meant to run under nothing at - all -- which `chat` is, being a conversation that ends when the person stops typing -- - and is what exempts it from being asked to confirm an unbounded run. A flow does not hold - itself to any of this: what holds a run to it is every session of every agent in it, so - this is a default the run reads and never a thing the flow implements. - - Args: - call: The function, when the decorator is written with no arguments at all. - name: What to call this one among the flows its directory holds, or "" for the one it - holds under the directory's own name. - about: One line saying what it does, defaulting to the first line of its docstring. - skills: The skills it works by that are somewhere else, one git URL apiece with an - optional `#`. What it keeps in its own `skills/` needs no declaring. - resumable: Whether it takes the state of the last run of it, and is handed a dict to - write the next run's into. - selectable: Whether to offer it in flow lists and the flow picker. False keeps an - internal composition callable by name without presenting it as a flow to start. - budget: What a run of it may spend by default, None for a flow with no opinion, and - `Allowance()` for one that means to run under nothing at all. - - Returns: - The function, unchanged but for what it now says about itself: a flow is called the way - it always was, and a decorator that wrapped it would put itself between the flow and - whatever reads its arguments. - """ - - def marks(said: Callable[P, T]) -> Callable[P, T]: - setattr( - said, - _SAID, - Flow( - name=name, - about=about or _first(said.__doc__), - skills=tuple(skills), - resumable=resumable, - selectable=selectable, - budget=budget, - ), - ) - return said - - return marks if call is None else marks(call) - - -def _first(said: str | None) -> str: - """The first line of a docstring, which is what a flow says about itself in a list. - - "" for a docstring that is blank, which is a docstring somebody left room in rather than - a flow to refuse: a flow says what it does or it does not. - """ - lines = (said or "").strip().splitlines() - return lines[0].strip() if lines else "" diff --git a/src/hmz/_legacy_flows/agent.py b/src/hmz/_legacy_flows/agent.py deleted file mode 100644 index 5eac8437..00000000 --- a/src/hmz/_legacy_flows/agent.py +++ /dev/null @@ -1,695 +0,0 @@ -"""What a flow drives: an agent, the conversations it opens, and the person at the prompt. - -Interfaces and nothing else. A flow is written against what it may ask of an agent -- a turn, a -session, a goal, what the turn cost -- and not against the class that answers: which CLI is behind -it, how a turn is spelled to that CLI, where its logs go and how it is stopped are all -:mod:`hmz.coganchor.agents`'s business and none of the flow's. So the contract is written here, -where a flow can import it beside the mark that makes it a flow, and the drivers implement it. - -Structurally rather than by inheritance, which is what keeps the arrow pointing one way: a -flow names what it drives, and a driver is written without ever naming a flow. What holds the -two together is checked where a type checker can see it, at the foot of this file, so that a -driver which stops answering to this reads as a driver to correct rather than as a flow that -fails on its first turn. - -A flow never makes one of these. The agents are chosen where the flow is started -- a command -line, the flow picker, another flow handing over its own -- and arrive as the tuple the flow -declared:: - - @flow - def run(agents: tuple[Agent, Agent], task: str) -> None: - builder, reviewer = agents -""" - -from __future__ import annotations - -from typing import TYPE_CHECKING, Any, ClassVar, Protocol, overload - -if TYPE_CHECKING: - import os - from collections.abc import Callable, Iterable, Iterator, Sequence - - from pydantic import BaseModel - - from hmz.coganchor.agents import ( - AgentConfig, - Board, - Budget, - Event, - Hooks, - Moment, - Question, - Tool, - Usage, - ) - from hmz.coganchor.agents.base import Journal - from hmz.coganchor.agents.skills import Loaded - from hmz.coganchor.machines import MachineConfig - -__all__ = ["Agent", "Driven", "Person", "Session"] - - -class Session(Protocol): - """One conversation with one agent, kept alive across turns. - - The first turn opens it and every later one resumes it, so the agent still has the earlier - turns in context. Letting go of it is how a flow forgets: the next one starts from nothing, - which is what a Ralph loop is made of. - """ - - #: Whether this backend can be held to a shape rather than asked to keep to one. A flow - #: that reads an answer as an object gets one either way; this is how sure it can be. A - #: fact of the backend, so it is written on the class -- a stand-in for one says it the - #: same way, `shapes: ClassVar[bool] = True`. - shapes: ClassVar[bool] - - #: Whether this backend can be given a tool the flow wrote. A fact of the backend, said - #: the same way, so a flow that offers a callback can be written to ask first rather than - #: to catch the refusal. - takes_tools: ClassVar[bool] - - #: Whether a turn of this backend can be talked to while it is running, which is what - #: :meth:`interject` reaches for. A fact of the backend, said the same way, so a flow that - #: means to steer a turn asks before it does rather than catching the refusal from a - #: backend that took the whole prompt up front. - steers: ClassVar[bool] - - #: Whether a turn of this backend can be told to say what it is reaching for while the - #: arguments are still being written, rather than once the whole of the call has arrived. - #: A fact of the backend, said the same way, and the one that says what a long silence - #: means: where this is false a turn writing a large file says nothing until it has - #: finished writing it, so a flow watching for signs of life knows not to read that as a - #: turn that has wedged. Whether a given agent is told to is its config's -- Claude Code's - #: `partial_messages`, on unless a flow says otherwise. - narrates: ClassVar[bool] - - @property - def forks(self) -> bool: - """Whether this backend can carry this conversation into a second one. - - A fact of the backend rather than of this conversation, but asked here because it is - here it is acted on: a flow that means to branch asks before it does, rather than - catching the refusal on a session it has already spent an hour filling. - """ - ... - - def fork(self) -> Session: - """A second conversation carrying this one's history, and its own from here on. - - Which is how a flow tries two ways out of an hour of work without paying for the hour - twice:: - - careful, quick = session.fork(), session.fork() - - The backend's own fork does the carrying, so the child knows exactly what this - conversation knew when it was made and nothing either of them is told afterwards. Its - own id, its own spending, its own line in the run's record: nothing spent here is - counted there. - - Not `Agent.clone`, which is the other half of the same idea under the other word: an - agent is structure, so its clone knows nothing; a session is history, so its fork - knows everything this one knows. - - The child must be used before this conversation is given another turn -- the fork is - the child's first turn, and one taken later would branch from somewhere nobody chose. - A child driven after that is refused rather than cut from the wrong place. - - Returns: - The new conversation, which opens with the backend on its first turn -- so a fork - nobody uses costs nothing at all. - - Raises: - NotImplementedError: If this backend has no fork of its own. `forks` is how to ask - first; a second handle on the one conversation is not offered instead. - RuntimeError: If no turn has landed here yet, so there is nothing to carry, or if - this conversation has moved on since the fork was asked for. - """ - ... - - @property - def id(self) -> str: - """What the backend calls this conversation, once a turn has landed in it. - - Raises: - RuntimeError: If no turn has landed yet, so the backend has not named it. - """ - ... - - @property - def named(self) -> str | None: - """The same name, or None while the backend has not said one.""" - ... - - @property - def cwd(self) -> str: - """The directory this conversation works in, as whoever is watching would name it.""" - ... - - @property - def effort(self) -> str: - """How hard the next turn of this conversation is to think.""" - ... - - @effort.setter - def effort(self, effort: str) -> None: ... - - @property - def skills(self) -> tuple[str, ...]: - """The flow's skills this conversation carries, by name, in the flow's own order.""" - ... - - @property - def budget(self) -> Budget | None: - """What each turn of this conversation may spend before it is cut off. - - What its agent was set up with, unless this conversation has been told otherwise -- - which a flow may do while the session is running, and which is where a loop watching - what a round costs is when it decides the next one is to be shorter:: - - session.budget = Budget(seconds=90, when="immediately", then="end") - - Per turn: every turn starts with the whole of it. None is a turn that runs until it - is done, and an empty `Budget()` is how one conversation opts out of the budget its - agent carries. The turn already under way keeps the budget it opened with. - """ - ... - - @budget.setter - def budget(self, budget: Budget | None) -> None: ... - - def interrupt(self, *, why: str) -> None: - """Cuts the turn now running off, wherever it has got to. - - Not the same as stopping the agent, which prevents the *next* turn: this one ends - the turn in flight, and the turn still finishes the way every turn finishes -- on one - answer, holding what the agent got as far as saying. A session with no turn running - is left alone. - - A backend whose turn is held somewhere shared -- an app server serving every session - of an agent at once -- stops at the next answer instead of being taken down, since - cutting one turn off must not end every other conversation on it. - - Args: - why: What it was cut off for, which is what whoever is watching the run reads. - """ - ... - - @property - def tools(self) -> tuple[Tool, ...]: - """The flow's own callbacks this conversation is putting in front of the agent.""" - ... - - def offers(self, tools: Iterable[Tool] | None) -> None: - """Says which callbacks of the flow's the agent may reach for, from the next turn on. - - A callback is a function of the flow's own, and the agent reaching for it is that - function running in the flow's process -- so a tool whose callback runs another flow - is an agent that can call a flow:: - - session.offers([Tool(name="review", about="have the reviewer read a file", - takes=Reviewing, call=lambda said: reviewer(said.path))]) - - Args: - tools: The callbacks, or None to take back whatever this conversation was offering. - - Raises: - NotImplementedError: On a backend with no way of being given a tool it was not - shipped with, which `takes_tools` says of the class beforehand. - """ - ... - - def loads(self, skills: Iterable[str] | None) -> None: - """Says which of the flow's skills this conversation carries from its next turn on. - - The one thing about what an agent works by that changes while it is working, and it - is the conversation's rather than the agent's: an agent is what it was made as, and a - conversation is a thing that gets somewhere. A loop that has finished reading and - started writing says so here. - - Args: - skills: The names to carry, or None for every one the flow brought. A name the flow - does not bring is ignored rather than refused, so a session asking for one a fork - of the flow no longer has carries the rest. - """ - ... - - @overload - def __call__(self, prompt: str, *, suppress: bool = False) -> str: ... - - @overload - def __call__[T: BaseModel]( - self, prompt: str, *, suppress: bool = False, schema: type[T] - ) -> T | None: ... - - def __call__[T: BaseModel]( - self, prompt: str, *, suppress: bool = False, schema: type[T] | None = None - ) -> str | T | None: - """Sends one turn, opening the session on the first call and resuming it after.""" - ... - - @overload - async def aturn(self, prompt: str, *, suppress: bool = False) -> str: ... - - @overload - async def aturn[T: BaseModel]( - self, prompt: str, *, suppress: bool = False, schema: type[T] - ) -> T | None: ... - - async def aturn[T: BaseModel]( - self, prompt: str, *, suppress: bool = False, schema: type[T] | None = None - ) -> str | T | None: - """The same turn, awaited: `await session.aturn(prompt)`.""" - ... - - def pursue(self, objective: str, *, suppress: bool = False) -> str: - """Runs the backend's own goal feature in this conversation, until it stops.""" - ... - - async def apursue(self, objective: str, *, suppress: bool = False) -> str: - """The same goal, awaited.""" - ... - - def stream( - self, prompt: str, *, schema: type[BaseModel] | None = None - ) -> Iterator[Event]: - """Sends one turn, saying what the agent says as it says it.""" - ... - - def spent(self) -> Usage: - """What this conversation has cost so far, by the kind of token it went on.""" - ... - - def rate(self, over: float = ...) -> Usage: - """How fast it is spending, by kind, over the last stretch of it.""" - ... - - def juice(self, over: float = ...) -> float: - """What an average turn of the model came out with, over that stretch.""" - ... - - def interject(self, text: str) -> None: - """Says something into the turn now running, for a backend that can be told.""" - ... - - def steering(self, text: str, ticket: str = "") -> str: - """Notes something said to this conversation that the agent has not yet said it has.""" - ... - - def took(self, ticket: str) -> str | None: - """What was said under that ticket, once the agent has said it has it.""" - ... - - def unsteered(self, text: str) -> None: - """Takes back something said that the agent will now never hear.""" - ... - - def close(self) -> None: - """Ends this conversation, and takes away whatever it put in the workspace.""" - ... - - -class Agent(Protocol): - """A coding agent behind a uniform interface: structure only, and no history. - - An agent says which model to run at which effort, and is one agent apart from that: a flow - that reviews its own work runs two of them at one configuration, and they are not the same - agent. The conversation lives in the :class:`Session` it opens, so a flow decides for - itself whether turns share context -- a fresh session per turn is a Ralph loop, one session - across turns is a stateful one. - """ - - #: The moments of a turn a hook may be hung on here. A flow that needs one only some - #: backends reach says so where it declares the place, and is then given an agent that - #: runs it rather than finding out from a hook that raised hours in. - moments: ClassVar[frozenset[Moment]] - - #: Whether this backend has a goal feature of its own, which is what :meth:`pursue` - #: reaches for. A flow that runs an agent under one says so where it declares the place. - #: - #: Both are facts of the backend rather than of any one agent, so both are written on the - #: class -- and a stand-in written for a test says them the same way, annotation and all: - #: `pursues: ClassVar[bool] = True`. - pursues: ClassVar[bool] - - #: Where the run this agent is being driven in is written down, or None for an agent - #: nobody is keeping a record of -- one driven from a test, or by a flow that was called - #: from nothing. Set by whatever started the run rather than by the flow. - epic: Journal | None - - @property - def id(self) -> str: - """What this agent is called, which is what a trace groups its sessions under.""" - ... - - @property - def backend(self) -> str: - """Which CLI is behind it, by the name a command line calls that CLI.""" - ... - - @property - def config(self) -> AgentConfig: - """The model, effort, account and permission every session of this agent runs at.""" - ... - - @property - def hooks(self) -> Hooks: - """What is hung on the moments of this agent's turns.""" - ... - - @property - def sessions(self) -> Sequence[Session]: - """The conversations opened on it and still held by someone, oldest first.""" - ... - - @property - def opened(self) -> list[str]: - """The backend's id for each conversation it has opened, in the order it opened them.""" - ... - - @property - def stopped(self) -> bool: - """Whether it has been told to take no further turn.""" - ... - - @property - def goals_enabled(self) -> bool: - """Whether its backend's own goal feature is switched on for it.""" - ... - - @property - def loaded(self) -> tuple[Loaded, ...]: - """The skills the flow it is being driven by mounts onto every session it opens.""" - ... - - @property - def effort(self) -> str: - """How hard its next turn is to think, in that backend's own word for it.""" - ... - - @effort.setter - def effort(self, effort: str) -> None: ... - - def new(self, cwd: str | os.PathLike[str] | None = None) -> Session: - """Opens a conversation, which stays unopened with the backend until its first turn.""" - ... - - @overload - def __call__( - self, - prompt: str, - *, - suppress: bool = False, - cwd: str | os.PathLike[str] | None = None, - ) -> str: ... - - @overload - def __call__[T: BaseModel]( - self, - prompt: str, - *, - suppress: bool = False, - schema: type[T], - cwd: str | os.PathLike[str] | None = None, - ) -> T | None: ... - - def __call__[T: BaseModel]( - self, - prompt: str, - *, - suppress: bool = False, - schema: type[T] | None = None, - cwd: str | os.PathLike[str] | None = None, - ) -> str | T | None: - """Runs one turn in a session of its own, and keeps nothing.""" - ... - - @overload - async def aturn( - self, - prompt: str, - *, - suppress: bool = False, - cwd: str | os.PathLike[str] | None = None, - ) -> str: ... - - @overload - async def aturn[T: BaseModel]( - self, - prompt: str, - *, - suppress: bool = False, - schema: type[T], - cwd: str | os.PathLike[str] | None = None, - ) -> T | None: ... - - async def aturn[T: BaseModel]( - self, - prompt: str, - *, - suppress: bool = False, - schema: type[T] | None = None, - cwd: str | os.PathLike[str] | None = None, - ) -> str | T | None: - """The same turn, awaited: `await agent.aturn(prompt)`.""" - ... - - def pursue( - self, - objective: str, - *, - suppress: bool = False, - cwd: str | os.PathLike[str] | None = None, - ) -> str: - """Runs a goal in a session of its own, and keeps nothing.""" - ... - - async def apursue( - self, - objective: str, - *, - suppress: bool = False, - cwd: str | os.PathLike[str] | None = None, - ) -> str: - """The same goal, awaited.""" - ... - - def batch_new( - self, count: int, cwd: str | os.PathLike[str] | None = None - ) -> Sequence[Session]: - """Opens as many conversations as it is asked for, at once.""" - ... - - @overload - def batch( - self, - prompts: Sequence[str], - *, - suppress: bool = False, - at_once: int = 0, - cwd: str | os.PathLike[str] | None = None, - ) -> list[str]: ... - - @overload - def batch[T: BaseModel]( - self, - prompts: Sequence[str], - *, - suppress: bool = False, - schema: type[T], - at_once: int = 0, - cwd: str | os.PathLike[str] | None = None, - ) -> list[T | None]: ... - - def batch[T: BaseModel]( - self, - prompts: Sequence[str], - *, - suppress: bool = False, - schema: type[T] | None = None, - at_once: int = 0, - cwd: str | os.PathLike[str] | None = None, - ) -> list[Any]: - """Runs many turns at once, each in a session of its own, and keeps none of them.""" - ... - - @overload - async def abatch( - self, - prompts: Sequence[str], - *, - suppress: bool = False, - at_once: int = 0, - cwd: str | os.PathLike[str] | None = None, - ) -> list[str]: ... - - @overload - async def abatch[T: BaseModel]( - self, - prompts: Sequence[str], - *, - suppress: bool = False, - schema: type[T], - at_once: int = 0, - cwd: str | os.PathLike[str] | None = None, - ) -> list[T | None]: ... - - async def abatch[T: BaseModel]( - self, - prompts: Sequence[str], - *, - suppress: bool = False, - schema: type[T] | None = None, - at_once: int = 0, - cwd: str | os.PathLike[str] | None = None, - ) -> list[Any]: - """The same batch, awaited.""" - ... - - def spent(self) -> Usage: - """What every conversation of this agent has cost, by the kind of token.""" - ... - - def rate(self, over: float = ...) -> Usage: - """How fast it is spending, by kind, over the last stretch of the run.""" - ... - - def juice(self, over: float = ...) -> float: - """What an average turn of the model came out with, over that stretch.""" - ... - - def stop(self) -> None: - """Has it take no further turn, and ends the one it is taking.""" - ... - - def watch(self, listener: Callable[[Agent, Session | None, Event], None]) -> None: - """Has everything its turns say reach `listener` as they say it.""" - ... - - def clone( - self, - *, - config: AgentConfig | None = None, - name: str | None = None, - skills: Iterable[Loaded] | None = None, - ) -> Agent: - """Another agent of this one's backend, differing in what this names and nothing else. - - The one way a flow has of getting an agent that is not quite the one it was handed, - and it is a second agent rather than this one changed:: - - careful = agent.clone(config=replace(agent.config, effort="max")) - - Everything an agent *is* is settled where it is made, so this is where it is settled - for the new one and there is nowhere it can be said again. What a run puts on an agent - rather than sets it up with does not come across: the clone has opened no conversation, - spent nothing, is watched by nobody and is being written down nowhere -- and it is not - one of the agents the run was started with, so what it does is its own. - - Args: - config: What every session of it runs at, or None for this agent's own. - name: What to call it, or None for one nothing else answers to: two agents are two - agents, and a trace that read a clone as its original would read a comparison of - two efforts as one agent changing its mind. - skills: The flow's skills it carries, or None for the ones this agent carries. - - Returns: - The new agent. - """ - ... - - def asked(self, question: Question) -> str | None: - """Puts something a turn stopped to ask to whoever is driving this agent.""" - ... - - def prompted(self) -> str | None: - """Waits for the next thing to say to it, for a flow that is a conversation.""" - ... - - -class Driven(Agent, Protocol): - """An agent as whoever hands it to a flow holds it: everything above, and settling it. - - The line between the two is who is entitled to say what an agent is. A flow is handed - agents and drives them; what each of them runs, where its turns land, what it is called - and which of the flow's skills it carries are answers somebody already gave -- at a - prompt, on a command line, in a settings file -- and a flow that could change them would - be a flow rewriting the choice it was started with. So they are not on :class:`Agent`, - and a flow that wants one set up differently makes one with :meth:`Agent.clone`. - - They are still on something, because somebody does settle them: `hmz.runtime.runner` - before the first turn, `hmz.runtime.flowing.driving` around a flow that called another, - and the interface when somebody watching a run says this agent is to go on as something - else. That is this. - """ - - def rename(self, name: str) -> None: - """Calls it something else, which is what a trace will group its sessions under.""" - ... - - def runs_on(self, machine: MachineConfig | None) -> None: - """Points its turns at a machine, or at this one, before any of them has landed. - - Settled by whatever hands the agent to a flow rather than by the flow: where an agent - works is what the flow's own declaration said, and a flow that moved one afterwards - would be undoing what it declared. - """ - ... - - def reconfigure(self, config: AgentConfig) -> None: - """Has its next session run at another model, effort, account or permission.""" - ... - - def loads(self, skills: Iterable[Loaded]) -> None: - """Tells it which of a flow's skills its sessions may carry from now on.""" - ... - - def disable_goals(self) -> None: - """Switches its backend's own goal feature off for the rest of the run.""" - ... - - -class Person(Agent, Protocol): - """The person at the prompt, driven as an agent so that a flow can talk to them. - - Everything an :class:`Agent` is, and the answers are typed rather than generated. A flow - declares one where it wants to be able to ask -- `agents: tuple[Agent, Person]` -- and it - is made rather than chosen: nobody picks a model for the person, so nothing that starts a - flow asks about this place. Run where nobody is at a prompt, it answers with nothing, and - a flow written to stop when it is told nothing stops. - """ - - @property - def board(self) -> Board: - r"""What the flow and the person both write on, and neither waits at. - - The other half of talking to them. A question stops the turn until it is answered; - this stops nothing at all -- a handful of named lines kept beside the run and shown - where the run is shown, which the flow reads and writes whenever it likes and the - person changes whenever they like:: - - while waiting := person.board.get("todo").splitlines(): - person.board.put("doing", waiting[0]) - builder(waiting[0]) - person.board.put("todo", "\n".join(waiting[1:])) - - A line may be one side's alone -- `whose="flow"` for a note the person is to read and - not rewrite, `whose="user"` for one the flow is to read and not -- and the other side - is refused where it writes rather than quietly ignored. - """ - ... - - -if TYPE_CHECKING: - from hmz.coganchor.agents import AgentBase, HumanAgent, SessionBase - - # : The one line that says :mod:`hmz.coganchor.agents` answers to the interfaces above. Written - # as : an assignment rather than as inheritance because the arrow points the other way: a : flow - # names what it drives, and a driver is written without ever naming a flow -- so : what joins - # the two is checked here, where a type checker reads it, and a driver that : stops answering to - # this reads as a driver to correct rather than as a flow that fails : on its first turn. - _implemented: tuple[type[Agent], type[Driven], type[Session], type[Person]] = ( - AgentBase, - AgentBase, - SessionBase, - HumanAgent, - ) diff --git a/src/hmz/_legacy_flows/atlas.py b/src/hmz/_legacy_flows/atlas.py deleted file mode 100644 index ac03463f..00000000 --- a/src/hmz/_legacy_flows/atlas.py +++ /dev/null @@ -1,354 +0,0 @@ -"""What an atlas is written in: the marks a body declares its graph with. - -A flow is a Python file that may branch any way it likes, and the one thing nothing can ask -it is what it is about to do. An atlas is the other bargain: a narrower Python, whose entry -point is read rather than run, and whose shape is therefore a graph that exists before -anything does. This is the half of that an atlas author writes -- the marks. What they are -compiled *to* is :mod:`hmz.runtime.flowing.prophecy`, and the compiling itself is -:mod:`hmz.runtime.flowing.prophesying`: neither is a thing an atlas names, so neither is a -thing this holds. - -An atlas is a flow. It is marked, found, named, listed and run the way every other flow is, -so nothing that already knows what a flow is has to learn a second thing:: - - from hmz._legacy_flows import Agent, atlas, logic, mind - from pydantic import BaseModel - - class Agents(NamedTuple): - writer: Agent - reviewer: Agent - - class Draft(BaseModel): - model_config = {"extra": "forbid"} - text: str - - class Verdict(BaseModel): - model_config = {"extra": "forbid"} - done: bool - - @mind - def write(agent: Agent, task: str) -> Draft: ... - - @logic - def judge(said: Draft) -> Verdict: ... - - @atlas - def run(agents: Agents, task: str) -> None: - draft = write(agents.writer, task) - verdict = judge(draft) - while not verdict.done: - draft = write(agents.writer, task) - -There are two kinds of ordinary node and one kind that is a whole flow. A `mind` is a turn: -real work by a real agent, handed the agent the call site names. A `logic` is a Python -function: no agent, no turn, and a decision anything can read. An atlas called by another -atlas is a supernode -- one node from outside, one prophecy from within. - -A mind has one way out and a logic may have several. That is the whole of why the two are -told apart: a branch is a decision, and a decision nothing but a model made is a decision no -reading of the flow can state. So the node a branch hangs off is a logic node, and what a -model said reaches a branch by being read by one. - -A mark marks and does not wrap, the way :func:`~hmz._legacy_flows.flow` does and for the same -reason: a body is read rather than run, so what the mark said has to travel on the function, -where the reading will find it. -""" - -from __future__ import annotations - -from dataclasses import dataclass -from typing import TYPE_CHECKING, Literal, overload - -if TYPE_CHECKING: - from collections.abc import Callable, Iterable - -__all__ = [ - "ATLAS", - "MARKED", - "Atlas", - "Kind", - "Marked", - "Sub", - "atlas", - "logic", - "mind", - "sub", -] - -#: What a node is: a turn taken by an agent, a Python function, or a whole atlas of its own. -#: The first two are what a prophecy is made of, and the third is what one prophecy is made of -#: another by, which is what a supernode is. -type Kind = Literal["mind", "logic", "atlas"] - -#: Where a marked node keeps what its mark said, and where an atlas keeps that it is one. On -#: the function rather than in a table, for the reason `flow` puts it there: a file is read -#: by running it, and a mark that travels with the thing it describes is a mark there is only -#: one place to look for. -MARKED = "__humanize_node__" -ATLAS = "__humanize_atlas__" - - -@dataclass(frozen=True, slots=True) -class Atlas: - """That a function is an atlas, and what the mark said beyond what `Flow` holds. - - An atlas carries this as well as the :class:`~hmz._legacy_flows.Flow` every flow carries, so - that - everything which already reads flows goes on reading this one, and only what compiles it - has to know the difference. - - Attributes: - name: What it is called inside its own file, which is the half after the colon, and "" - for the one the file holds under its own name. The same name `flow` takes, and the - same name a supernode of another file is reached by. - """ - - name: str = "" - - -@dataclass(frozen=True, slots=True) -class Marked: - """What `mind` or `logic` marked a function with. - - Attributes: - kind: Which of the two it is. A mind takes a turn and has one way out; a logic is - Python and may have several. - rerun: Whether a run picked up again runs this node again where the last run was - stopped inside it. True is what a node says by saying nothing: work that was cut off - partway is work that was not done. False is for a node that has had its effect by - the time it can be interrupted -- and such a node answers with nothing, since a run - stepping past it has no answer of its to carry on with. - """ - - kind: Kind - rerun: bool = True - - -@dataclass(frozen=True, slots=True) -class Sub: - """An atlas of another file, as the atlas reaching for it names it. - - Bound at the top of a file and called in a body, which is the one way an atlas reaches a - flow that is not beside it:: - - review = sub("official/review") - - Never called: an atlas's body is read rather than run, and what runs is the prophecy the - reading compiled. Calling one is therefore an atlas being run some way this module knows - nothing about, and says so rather than doing something surprising. - - Attributes: - named: The flow, by the name `-f` takes. - """ - - named: str - - def __call__(self, *args: object, **kwargs: object) -> object: # noqa: ARG002 - """Refuses: an atlas's body is compiled, and the prophecy is what runs. - - Args: - args: Whatever the call was written with, which is read where it is written. - kwargs: The same. - - Raises: - TypeError: Always. A supernode is run by the prophecy around it, and a body that ran - would be an atlas being run as though it were an ordinary flow. - """ - raise TypeError( - f"{self.named} is a supernode: an atlas's body is compiled rather than run, " - "so nothing calls this outside the prophecy it was read into" - ) - - -def sub(named: str) -> Sub: - """Names the atlas one supernode is, for a body to call it by. - - The counterpart of :func:`~hmz._legacy_flows.load`, and the only one an atlas has: `load` - answers - with a flow that may be anything, and an atlas that called one would be a prophecy with a - hole where a node should be. So an atlas reaches another atlas, by the name `-f` takes, - and reaches nothing else. - - Args: - named: The flow, by the name `-f` takes -- `official/review`, `local/triage:pass`. - - Returns: - Something for a body to call, which nothing ever calls: it is read where it is written, - and the atlas it names is compiled into the prophecy reading it. - """ - return Sub(named) - - -@overload -def mind[**P, T](call: Callable[P, T], /) -> Callable[P, T]: ... - - -@overload -def mind[**P, T]( - *, rerun: bool = True -) -> Callable[[Callable[P, T]], Callable[P, T]]: ... - - -def mind[**P, T]( - call: Callable[P, T] | None = None, /, *, rerun: bool = True -) -> Callable[P, T] | Callable[[Callable[P, T]], Callable[P, T]]: - """Marks a function as a node an agent takes a turn in -- the work itself. - - A mind is handed the agent the call site named and whatever else flows into it, and - answers with a shape:: - - @mind - def write(agent: Agent, task: str) -> Draft: - return agent(f"draft this: {task}", schema=Draft) - - It has exactly one way out. What a model said is not a decision until something read it, - so a branch is hung off a logic node and never off this: a prophecy that branched on a - turn would be a prophecy whose shape is whatever the model happened to say. - - Args: - call: The function, when the mark is written with no arguments at all. - rerun: Whether a run picked up again runs this node again where the last one stopped - inside it, which is what a node says by saying nothing. - - Returns: - The function, unchanged but for what it now says about itself. - """ - return _noded("mind", call, rerun=rerun) - - -@overload -def logic[**P, T](call: Callable[P, T], /) -> Callable[P, T]: ... - - -@overload -def logic[**P, T]( - *, rerun: bool = True -) -> Callable[[Callable[P, T]], Callable[P, T]]: ... - - -def logic[**P, T]( - call: Callable[P, T] | None = None, /, *, rerun: bool = True -) -> Callable[P, T] | Callable[[Callable[P, T]], Callable[P, T]]: - """Marks a function as a node that is Python -- the deciding, the counting, the shaping. - - A logic drives no agent and takes no turn:: - - @logic - def judge(said: Draft) -> Verdict: - return Verdict(done=said.text.endswith(".")) - - It may have several ways out, which is what a branch in an atlas's body is: the value it - answered with is read by the `if` or the `while` that follows it, and each way out is one - answer to that reading. - - Args: - call: The function, when the mark is written with no arguments at all. - rerun: Whether a run picked up again runs this node again where the last one stopped - inside it, which is what a node says by saying nothing. - - Returns: - The function, unchanged but for what it now says about itself. - """ - return _noded("logic", call, rerun=rerun) - - -def _noded[**P, T]( - kind: Kind, call: Callable[P, T] | None, *, rerun: bool -) -> Callable[P, T] | Callable[[Callable[P, T]], Callable[P, T]]: - """Marks one function as a node, whichever of the two kinds it is. - - What the two marks share is the whole of how a decorator written bare and one written - with arguments are told apart, which is a protocol worth having in one place rather than - two: what differs between them is which kind it is, and what each says for itself. - - Args: - kind: Which kind of node the mark makes it. - call: The function, where the mark was written with no arguments at all. - rerun: Whether a run picked up inside it runs it again. - - Returns: - The function where there was one, and something to mark one where there was not. - """ - - def marks(said: Callable[P, T]) -> Callable[P, T]: - setattr(said, MARKED, Marked(kind, rerun=rerun)) - return said - - return marks if call is None else marks(call) - - -@overload -def atlas[**P, T](call: Callable[P, T], /) -> Callable[P, T]: ... - - -@overload -def atlas[**P, T]( - *, - name: str = "", - about: str = "", - skills: Iterable[str] = (), - selectable: bool = True, -) -> Callable[[Callable[P, T]], Callable[P, T]]: ... - - -def atlas[**P, T]( - call: Callable[P, T] | None = None, - /, - *, - name: str = "", - about: str = "", - skills: Iterable[str] = (), - selectable: bool = True, -) -> Callable[P, T] | Callable[[Callable[P, T]], Callable[P, T]]: - """Marks a function as an atlas: a flow whose body is a graph rather than a program. - - Everything :func:`~hmz._legacy_flows.flow` marks a flow with, this marks too -- the name, the - line it says about itself, the skills it works by, whether it is offered in a list -- so - an atlas is found, listed, chosen and run exactly as any other flow is. What it adds is - that the body is read instead of executed:: - - @atlas - def run(agents: Agents, task: str) -> None: - draft = write(agents.writer, task) - verdict = judge(draft) - - The body is a declaration. Each statement in it is one node; the branches between them - are the edges; and what actually runs is the prophecy that reading compiled, one node at a - time, which is what lets a run be picked up in the middle of one. - - An atlas can always be picked up again, and says so without being asked: a prophecy is a - list of nodes with an answer apiece, so what a run of one has done so far is something - the run itself writes down. Nothing in the body writes state and nothing is handed a - dict; a node that ran is a node whose answer was kept. - - An atlas that takes a shape rather than a task is a supernode and nothing else:: - - @atlas(name="review") - def review(agents: Agents, draft: Draft) -> Verdict: - ... - - Args: - call: The function, when the mark is written with no arguments at all. - name: What to call this one among the flows its file holds, or "" for the one it holds - under the file's own name. - about: One line saying what it does, defaulting to the first line of its docstring. - skills: The skills it works by that are somewhere else, one git URL apiece. - selectable: Whether to offer it in flow lists and the flow picker. - - Returns: - The function, unchanged but for what it now says about itself -- both marks, since an - atlas is a flow and everything that reads flows must go on reading this one. - """ - from . import flow - - def marks(said: Callable[P, T]) -> Callable[P, T]: - setattr(said, ATLAS, Atlas(name=name)) - return flow( - name=name, - about=about, - skills=skills, - resumable=True, - selectable=selectable, - )(said) - - return marks if call is None else marks(call) diff --git a/src/hmz/_legacy_flows/builtin/chat/__init__.py b/src/hmz/_legacy_flows/builtin/chat/__init__.py deleted file mode 100644 index d3c520aa..00000000 --- a/src/hmz/_legacy_flows/builtin/chat/__init__.py +++ /dev/null @@ -1,58 +0,0 @@ -"""Chat -- one agent, one session, and every line typed between turns is a turn of it. - -hmz exec -f chat -a claude/MODEL:high "what does this repository do?" - -Which is talking to a coding agent, with no loop around it: the flow does what it is told and -then waits to be told again. It is the flow the terminal interface opens on, so that saying -something is all it takes to start. A line typed while a turn is running is put into that turn -rather than becoming another, as it is under any flow. - -Two agents, then, and the second of them is you: saying something to the person is asking what -to say next, and what they answer is what they typed. Run from a command line, where nobody is -at a prompt, they answer with nothing and the flow does the one thing it was given and stops. - -The first turn is the one allowed to fail out loud. A conversation that could not be started --- an account refused, a model this backend will not run for it -- ends the run with what the -backend said about it, rather than answering with nothing and exiting as though the one thing -it was asked for had been done. Every turn after it is forgiving. - -Nothing of it is kept for a next run to pick up. What was said is the conversation, which the -backend that ran it logs turn by turn, and a session is opened rather than reopened -- so -starting this again is another conversation rather than the last one carried on. -""" - -from typing import NamedTuple - -from hmz._legacy_flows import Agent, Allowance, Person, flow - - -class Chat(NamedTuple): - """The two sides of a conversation.""" - - assistant: Agent - human: Person - - -# An allowance of nothing, written out rather than left unsaid: a conversation ends when -# the person stops typing, and there is no round of it they did not ask for. Written out is -# also what says so -- a flow that says nothing about its budget is asked to confirm that an -# unbounded run is what was meant, and a conversation is the one run where that question has -# an obvious answer and would be asked every time. -@flow(budget=Allowance()) -def run(agents: Chat, task: str) -> None: - # One session, so the turns are a conversation rather than a series of first turns. - conversation = agents.assistant.new() - said = task - opening = True - while said: - # The opening turn is not suppressed. A conversation whose first turn cannot run at - # all -- an account the backend refused, a model this one is not entitled to -- is a - # run to fail loudly; suppressed, it answers with nothing, which reads below as a - # conversation that is over, so the flow would end without a word and exit as though - # it had done what it was asked. Once a turn has landed the rest are forgiving, which - # is what a conversation is. - answered = conversation(said, suppress=not opening) - opening = False - # Saying that to the person is asking what to say next, and what they answer with is - # what they typed -- or nothing, which is a conversation that is over. - said = agents.human(answered) diff --git a/src/hmz/cli/__init__.py b/src/hmz/cli/__init__.py index 78d2e42c..11fa0b27 100644 --- a/src/hmz/cli/__init__.py +++ b/src/hmz/cli/__init__.py @@ -1,7 +1,7 @@ """``hmz`` -- the whole command line, over layers that have none of their own. hmz - hmz exec -f ralph_loop -a claude/MODEL:high "$(cat TASK.md)" + hmz exec -f ralph_loop -a coder=claude/MODEL:high -b cost=5 "$(cat TASK.md)" There is one command anybody types, and everything else humanize keeps is walked at the prompt: a listing with a noun in it for every store would be a second interface to learn, and @@ -22,8 +22,8 @@ A command whose line takes a parser of its own has a module of its own here, so that reaching one of them costs nothing for the others -- which is what `anchor.py`, `cred.py`, `hook.py` and `tools.py`, the four under `hmz internal`, are. `exec` has none: the line it takes is -read by :func:`hmz.runtime.runner.flow_and_agents`, since the terminal interface starts a flow from -that same line. +read by :func:`hmz.runtime.runner.read_line`, since the terminal interface starts a flow from +the same parts. :mod:`hmz.cli.output` is the one module every command may reach: who is reading -- somebody at a terminal, or a program -- is one question rather than one per command, and it costs nothing @@ -89,11 +89,11 @@ def _prepare_textual_terminal( def _exec(argv: list[str]) -> int: - """Drives the flow named on the command line, on the agents it names. + """Runs the flow named on the command line, on what it names, to its return. What the run looks like while it happens is settled here rather than by each backend - teeing its own progress: watching the agents is what makes one run read as one run, - whichever CLIs it was given, and registering a watcher is what stops those tees. + teeing its own progress: watching every session is what makes one run read as one run, + whichever CLIs it was given, and watching one is what stops that tee. Args: argv: What followed the command name. @@ -101,7 +101,7 @@ def _exec(argv: list[str]) -> int: Returns: Zero, once the flow has returned. """ - from hmz.runtime import Hmz, telemetry + from hmz.runtime import Hmz, Refused, telemetry from .output import Out, Shown @@ -109,60 +109,51 @@ def _exec(argv: list[str]) -> int: # If it has been answered yes, and never otherwise: a run with nobody at a terminal is a # run with nobody to ask, and silence is not an answer. hmz.reports() - path, agents, task, config, budget, as_json = hmz.read(argv) + line = hmz.read(argv) # Only now that the line is known to name a flow: `--help` has already exited inside the - # reading above, and a line that runs nothing must not pay for the drivers to be loaded - # so that this can learn the name of what a stopped run raises -- nor for what reads a - # flow, so that it can learn the name of what a line naming none is refused with. - from hmz.coganchor.agents import Stopped - from hmz.runtime.flowing import NotAFlow - - with Out(as_json=as_json) as out, Shown(out) as shown: - # The agents the line named, and not whatever else the flow turns out to drive: a - # flow whose other side is the person drives one more, and with nobody at a prompt - # that one answers nothing to every turn it is given. Rows saying so would be the - # only thing on the terminal that is about humanize rather than about the run. - shown.watches(agents) + # reading above, and a line that runs nothing must not pay for the flow API to learn the + # name of what a run that spent its budget raises. + from hmz.flows import BudgetExceeded + + with Out(as_json=line.as_json) as out, Shown(out) as shown: try: - running = hmz.run(path, agents, task, config, budget=budget) - except NotAFlow as error: - # A flow that is not there, or one that takes other agents than these, is a - # command line that was wrong before anything ran, so it exits as argparse's own - # rejections do. What the flow raises for itself is the flow's, and is left to - # say so itself. + # Nobody is at a prompt, so whoever is outside the run is away: a flow that asks + # the person anything is answered with nothing, or its schema's defaults. + running = hmz.run( + line.flow, + line.task, + agents=line.agents, + envs=line.envs, + params=line.params, + budget=line.budget, + resume=line.resume, + ) + except Refused as error: + # A flow that is not there, or one given other roles than it declares, is a + # command line that was wrong before anything ran, so it exits as argparse's + # own rejections do. What the flow raises for itself is the flow's. print(f"hmz exec: error: {error}", file=sys.stderr) raise SystemExit(2) from error - # Said and then run, never asked. The interface asks somebody to confirm a run that - # nothing will stop; a command line has nobody to ask, and refusing here would break - # every unattended flow there has ever been for the sake of a question nobody is - # there to answer. So what it can do is say so plainly, on the stream that is not - # the answer. - # - # Which cap cannot be read before what that leaves: a run whose only cap is one - # nothing here can read is both of these lines, and the first is the second's reason. + running.watch(shown.heard) + # Said and then run, never asked: a command line has nobody to ask, so what it can + # do is say so plainly, on the stream that is not the answer. if blind := running.unreadable(): - out.aside(f"hmz exec: {blind}, so that cap cannot stop this run") - if running.unwatched: - out.aside( - "hmz exec: nothing will stop this run -- it goes until it is stopped by " - "hand. `-c` with a `budget:` is how a cap that bites is set." - ) - # And what a place declared that its agent could not be told. Said for the same - # reason and on the same stream: a declaration that was dropped is a fact about how - # this run was set up, and one nobody was told about would be a setting that lied - # after all. - for line in running.unserved().splitlines(): - out.aside(f"hmz exec: {line}") + out.aside(f"hmz exec: {blind}") try: running.run() + except Refused as error: + # An environment that could not be reached, or one short of what its role + # needs, which is only known once it has been asked -- still before the flow ran. + print(f"hmz exec: error: {error}", file=sys.stderr) + raise SystemExit(2) from error except (KeyboardInterrupt, SystemExit): # Somebody stopping a run is not a run that went wrong. raise - except Stopped as why: + except BudgetExceeded as why: # Nor is a run that spent what it was allowed. It is the ordinary end of a - # budgeted loop -- a flow with no exit of its own runs until its allowance is - # gone, which is what having one is for -- so it is said in a line rather than - # reported as a crash and printed as a traceback nobody has anything to do about. + # budgeted loop -- a flow with no exit of its own runs until its budget is gone, + # which is what having one is for -- so it is said in a line rather than reported + # as a crash and printed as a traceback nobody has anything to do about. out.aside(f"hmz exec: stopped -- {why}") except BaseException as why: # Reported and then raised on exactly as it was: what a flow does when it fails @@ -392,6 +383,10 @@ def apart(session: Held) -> None: # the runtime itself, rather than being handed the answer by whatever it is holding. session.redrawn(lambda: app.call_from_thread(app.reattached)) session.stopping(lambda: app.call_from_thread(app.action_quit)) + # And what the interface says about the run it holds -- the flow, what it may spend and + # what it has spent -- which the daemon adds to its status. Read on the daemon's thread, + # off what the run keeps under its own locks rather than off the screen. + session.says(app.said) app.run() diff --git a/src/hmz/cli/output.py b/src/hmz/cli/output.py index 03b5681d..b6c6808e 100644 --- a/src/hmz/cli/output.py +++ b/src/hmz/cli/output.py @@ -35,7 +35,7 @@ from . import many if TYPE_CHECKING: - from collections.abc import Callable, Iterable + from collections.abc import Callable from types import TracebackType from typing import IO @@ -357,10 +357,10 @@ def spins(self, says: Callable[[], str] | None) -> None: class Shown: """A run of a flow as it happens, laid out for whoever is reading it. - Built on `AgentBase.watch`, which is the same stream the interface draws from -- so the - two readings of one run say the same things in the same order. Registering a watcher is - also what stops every driver teeing its own raw progress to stderr, so this replaces that - tee rather than being printed beside it. + Handed to `Run.watch`, which hands it to every session of the run -- the same stream the + interface draws from -- so the two readings of one run say the same things in the same + order. Watching a session is also what stops its CLI teeing its own raw progress to + stderr, so this replaces that tee rather than being printed beside it. """ def __init__(self, out: Out) -> None: @@ -398,15 +398,6 @@ def __exit__( """Takes it down again, however the run ended.""" self._out.spins(None) - def watches(self, agents: Iterable[AgentBase]) -> None: - """Has everything these agents' turns say reach this. - - Args: - agents: The agents the flow is being driven with. - """ - for agent in agents: - agent.watch(self.heard) - def heard( self, agent: AgentBase, session: SessionBase | None, event: Event ) -> None: diff --git a/src/hmz/daemon/__init__.py b/src/hmz/daemon/__init__.py index b6e9f45c..efbfb0e4 100644 --- a/src/hmz/daemon/__init__.py +++ b/src/hmz/daemon/__init__.py @@ -137,7 +137,7 @@ def status(self) -> dict[str, Any]: said = self.asked({"do": "status"}) if said.get("ok"): return said - return {**self._written(), "attached": 0, "flows": []} + return {**self._written(), "attached": 0, "flows": [], "calls": []} def detach(self) -> int: """Lets go of every terminal reading this run, leaving the run running. diff --git a/src/hmz/daemon/serve.py b/src/hmz/daemon/serve.py index c09cc5a9..effd16ef 100644 --- a/src/hmz/daemon/serve.py +++ b/src/hmz/daemon/serve.py @@ -575,23 +575,42 @@ def _status(self) -> dict[str, Any]: """ said: dict[str, Any] = dict(where.held(self._at)) said["attached"] = self.attached - said["flows"] = self._flows() + calls = self._calls() + said["flows"] = [one["ref"] for one in calls] + said["calls"] = calls hook = self._saying if hook is not None: with contextlib.suppress(Exception): said.update(hook()) return said - def _flows(self) -> list[str]: - """Every flow of the run being held here, oldest first, and none where none is. + def _calls(self) -> list[dict[str, Any]]: + """Every flow call of the run being held here, oldest first, and none where none is. - Answered rather than raised however it goes: a status nobody can read is worse than - one that says a run it could not ask about is running no flows. + The running tree, as JSON: each call's flow by its canonical ref, its name, how deep + it is, how long it has been going and the call that made it, by its place in this + list. Answered rather than raised however it goes: a status nobody can read is worse + than one that says a run it could not ask about is running no flows. """ + import time + with contextlib.suppress(Exception): from hmz.runtime import Hmz - return [one.flow for one in Hmz().flows.running()] + running = Hmz().flows.running() + now = time.monotonic() + at = {id(one): index for index, one in enumerate(running)} + return [ + { + "ref": one.ref, + "name": one.name, + "depth": one.depth, + "seconds": round(now - one.since, 3), + "id": one.id, + "parent": None if one.parent is None else at.get(id(one.parent)), + } + for one in running + ] return [] def _closing(self, selector: selectors.BaseSelector, one: socket.socket) -> None: diff --git a/src/hmz/_legacy_flows/builtin/__init__.py b/src/hmz/flows/builtin/__init__.py similarity index 81% rename from src/hmz/_legacy_flows/builtin/__init__.py rename to src/hmz/flows/builtin/__init__.py index ade16a10..577e78cb 100644 --- a/src/hmz/_legacy_flows/builtin/__init__.py +++ b/src/hmz/flows/builtin/__init__.py @@ -15,7 +15,8 @@ from here to there goes on answering to the name it always had. A directory of flows and nothing else, one directory apiece: the `__init__.py` that is the -flow, whatever it imports beside it, and the `skills/` it brings. A flowverse that is fetched -keeps its flows in a `flows/` directory, having a repository around them to keep out of the -way; these have none, being in the package, and are read where they stand. +flow, whatever it imports beside it, and the `skills/` it brings. Each is written against +:mod:`hmz.flows` like any other flow and imports nothing else of humanize's; what sets them +apart -- a flow here is handed every capability of its agent's harness -- is the runtime's to +say, in :func:`hmz.runtime.flowing.finding.resolved`. """ diff --git a/src/hmz/flows/builtin/chat/__init__.py b/src/hmz/flows/builtin/chat/__init__.py new file mode 100644 index 00000000..b1ebc03b --- /dev/null +++ b/src/hmz/flows/builtin/chat/__init__.py @@ -0,0 +1,124 @@ +"""Chat -- one agent, one session, and every line said back is the next turn of it. + + hmz exec -f chat -a assistant=claude/MODEL:high "what does this repository do?" + +Which is talking to a coding agent, with no loop around it: the flow does what it is told and +then waits to be told again. It is the flow the terminal interface opens on, so that saying +something is all it takes to start. + +Two agents, then, and the second of them is you: `human` is the outworlder, and saying +something to it is asking what to say next. Run from a command line, where nobody is at a +prompt, the outworlder is away and answers with nothing, so the flow does the one thing it was +given and stops. A question the agent stops to ask its user mid-turn is put to you the same +way, on a harness that asks. + +The agent is whatever harness was chosen, with everything that harness can do: `chat` declares +a plain `Agent` -- one allowed the web -- because it talks to any of them, and the runtime hands +the flows humanize ships the harness's full view. It is the one flow that runs with no budget +of its own -- a conversation ends when you stop talking, and the runtime runs it under +`Budget(cost=inf)`. + +The first turn is the one allowed to fail out loud. A conversation that could not be started +-- an account refused, a model this harness will not run -- ends the run with what the harness +said about it, rather than answering with nothing and exiting as though the one thing it was +asked for had been done. A turn after it that fails is said to you, and the conversation goes +on. + +Nothing of it is kept for a next run to pick up: a session is opened rather than reopened, so +starting this again is another conversation rather than the last one carried on. +""" + +from __future__ import annotations + +import contextlib +from typing import cast + +from hmz.flows import ( + Agent, + AgentCollection, + AskUserHookAgentMixin, + AskUserHookParams, + AskUserHookResult, + CapabilityNotGranted, + EnvCollection, + FlowContext, + FlowParams, + HarnessError, + HookFn, + LocalEnv, + Outworlder, + Permission, + PermissionKind, + Session, + flow, +) + + +class Assistant(Agent): + """Whichever agent was chosen, allowed the web as a person talking to one would expect.""" + + _permission = Permission(online=PermissionKind.ALL) + + +class Agents(AgentCollection): + """The two sides of a conversation.""" + + assistant: Assistant + human: Outworlder + + +class Envs(EnvCollection): + """Where the conversation happens: the workspace it was started in.""" + + workspace: LocalEnv + + +class Params(FlowParams): + """Nothing: a conversation is set up by what is said in it.""" + + +@flow(agents=Agents, envs=Envs, params=Params) +async def chat( + task: str, + *, + agents: Agents, + envs: Envs, + params: Params, # noqa: ARG001 -- a flow takes its params whether or not it has any + ctx: FlowContext, # noqa: ARG001 -- likewise its context +) -> None: + """Talks to one agent for as long as you keep answering it.""" + assistant, human = agents["assistant"], agents["human"] + here = envs["workspace"] + # One session, so the turns are a conversation rather than a series of first turns. + conversation = await assistant.spawn(env=here) + person = await human.spawn(env=here) + with contextlib.suppress(CapabilityNotGranted): + # A harness that stops to ask its user a question has it put to the person here; + # one that cannot ask has no such hook to hang, and there is nothing to put. + cast("AskUserHookAgentMixin", assistant).on_ask_user(_asking(human, person)) + said = task + opening = True + while said: + try: + answered = await assistant.run(said, session=conversation) + except HarnessError as failed: + if opening: + raise + answered = f"That turn could not be taken: {failed}" + opening = False + # Saying that to the person is asking what to say next, and what they answer with is + # what they typed -- or nothing, which is a conversation that is over. + said = await human.run(answered, session=person) + + +def _asking( + human: Outworlder, person: Session +) -> HookFn[AskUserHookParams, AskUserHookResult]: + """The hook that puts a question the agent asked to the person it is talking to.""" + + async def asked(params: AskUserHookParams) -> AskUserHookResult: + offered = f" ({' / '.join(params.options)})" if params.options else "" + said = await human.run(f"{params.question}{offered}", session=person) + return AskUserHookResult(answer=said or None) + + return asked diff --git a/src/hmz/runtime/__init__.py b/src/hmz/runtime/__init__.py index 7da47eef..dbeb9750 100644 --- a/src/hmz/runtime/__init__.py +++ b/src/hmz/runtime/__init__.py @@ -3,12 +3,12 @@ from hmz.runtime import Hmz hmz = Hmz() - hmz.run("chat", [], "say hello").run() + hmz.run("chat", "say hello", agents={"assistant": "claude/claude-haiku-4-5:low"}).run() The layer between what a flow says and the agents that do it. It finds the flow, hands it -the agents it declared, opens the epic a run is written into as it happens, remembers what -this workspace was set up with, and reads the whole of it back again -- as a trace, or as -one archive to send somewhere. +a driver for every role it declared, opens the epic a run is written into as it happens, +remembers what this workspace was set up with, and reads the whole of it back again -- as a +trace, or as one archive to send somewhere. :class:`Hmz` is the front door: one workspace and everything that can be done in it, composed out of the modules beside it in :mod:`hmz.runtime.doing`. A command line names it, and so does @@ -34,6 +34,7 @@ from hmz.runtime.doing.fallbacks import Fallbacks from hmz.runtime.doing.flows import Flows, Flowverses from hmz.runtime.doing.running import Run + from hmz.runtime.runner import Refused __all__ = [ "Accounts", @@ -42,6 +43,7 @@ "Flows", "Flowverses", "Hmz", + "Refused", "Run", ] @@ -54,6 +56,7 @@ "Flows": "hmz.runtime.doing.flows", "Flowverses": "hmz.runtime.doing.flows", "Hmz": "hmz.runtime.doing.core", + "Refused": "hmz.runtime.runner", "Run": "hmz.runtime.doing.running", } diff --git a/src/hmz/runtime/doing/core.py b/src/hmz/runtime/doing/core.py index a88e918e..999b3a62 100644 --- a/src/hmz/runtime/doing/core.py +++ b/src/hmz/runtime/doing/core.py @@ -22,19 +22,18 @@ if TYPE_CHECKING: import os - from collections.abc import Mapping, Sequence + from collections.abc import Iterable, Mapping - from pydantic import BaseModel - - from hmz.coganchor.agents import AgentBase - from hmz.coganchor.agents.allowance import Allowance from hmz.coganchor.backends import Profile + from hmz.flows import Budget, FlowParams from hmz.runtime.doing.accounts import Accounts from hmz.runtime.doing.epics import Epics from hmz.runtime.doing.fallbacks import Fallbacks from hmz.runtime.doing.flows import Flows, Flowverses from hmz.runtime.doing.running import Run - from hmz.runtime.runner import Runner + from hmz.runtime.flowing import AgentDriver, EnvDriver, OutworlderDriver + from hmz.runtime.flowing.specs import AgentSpec, EnvSpec + from hmz.runtime.runner import Line, Runner from hmz.runtime.settings import Settings __all__ = ["Hmz"] @@ -141,105 +140,133 @@ def reports(self) -> bool: return telemetry.start() - def read( - self, argv: list[str] - ) -> tuple[ - str, list[AgentBase], str, dict[str, Any] | None, Allowance | None, bool - ]: - """Reads an `hmz exec` line into a flow, the agents, the task, and the flow's setup. + def read(self, argv: list[str]) -> Line: + """Reads an `hmz exec` line: the flow, what each role is given, the params and budget. Args: argv: The line, as `hmz exec` takes it. Returns: - The flow's path, the agents to drive it with in the order the flow takes them, the - task, what to set the flow up with, what the run may spend, and whether the line - asked for the run to be written for a program rather than for a person. + The line, read. Nothing is loaded: whether the flow takes what it names is asked + of it by :meth:`runner`. Raises: - SystemExit: If the line does not name a flow and an agent apiece, as argparse - rejects it. + SystemExit: If the line is not one argparse accepts, or an `-a`, `-e`, `-p` or `-b` + on it cannot be read. """ - from hmz.runtime.runner import flow_and_agents + from hmz.runtime.runner import read_line - return flow_and_agents(argv) + return read_line(argv) def runner( self, flow: str | os.PathLike[str], - agents: Sequence[AgentBase], - config: BaseModel | dict[str, Any] | None = None, - resume: str | os.PathLike[str] | None = None, - container: str = "", - budget: Allowance | Mapping[str, Any] | None = None, + *, + agents: Mapping[str, str | AgentDriver] | Iterable[AgentSpec] = (), + envs: Mapping[str, str | EnvDriver] | Iterable[EnvSpec] = (), + params: Mapping[str, Any] | FlowParams | None = None, + budget: Budget | Mapping[str, Any] | None = None, + resume: bool | str | os.PathLike[str] = False, ) -> Runner: - """Loads a flow and hands it the agents it was written for. + """Loads a flow and opens a driver for every role it is given, checking all of it. Args: - flow: The Python file the flow is written in, or the name it is offered under. - agents: The agents to hand it, as many as it declares. - config: What it was set up with, for a flow that says it can be. - resume: The run to pick up from, for a flow that says it can be picked up. - container: The image to run the whole of it in, or "" for this machine. - budget: What the run may spend, or None for whatever the flow says. + flow: The flow, by the name it is offered under, a path, or a ref. + agents: What each agent role runs, by role -- an `-a` spec after `=`, or a + driver -- or the specs a line read. + envs: What each environment role is, likewise with `-e`. + params: The flow's params, or None for its defaults. + budget: What the run may spend; only a flow humanize ships runs without one. + resume: Whether to pick up the newest run of it here, or the epic to pick up. Returns: - The flow, loaded, with the agents it drives in hand. + The flow, loaded, with its drivers in hand and nothing started. Raises: - NotAFlow: If the flow is not there, is not a flow, or takes other agents than these. + Refused: If the flow is not there, or is given what it does not declare, or is not + given what it needs -- before anything runs. """ from hmz.runtime.runner import Runner return Runner( - flow, agents, config, resume=resume, container=container, budget=budget + flow, + agents=agents, + envs=envs, + params=params, + budget=budget, + resume=resume, + workspace=self._workspace, ) def run( self, flow: str | os.PathLike[str], - agents: Sequence[AgentBase], task: str, - config: BaseModel | dict[str, Any] | None = None, - resume: str | os.PathLike[str] | None = None, - container: str = "", - budget: Allowance | Mapping[str, Any] | None = None, + *, + agents: Mapping[str, str | AgentDriver] | Iterable[AgentSpec] = (), + envs: Mapping[str, str | EnvDriver] | Iterable[EnvSpec] = (), + params: Mapping[str, Any] | FlowParams | None = None, + budget: Budget | Mapping[str, Any] | None = None, + resume: bool | str | os.PathLike[str] = False, + outworlder: OutworlderDriver | None = None, ) -> Run: """A run of one flow, loaded and ready to be started. Args: - flow: The Python file the flow is written in, or the name it is offered under. - agents: The agents to hand it, as many as it declares. - task: What the flow is to have them do. - config: What it was set up with, for a flow that says it can be. - resume: The run to pick up from, for a flow that says it can be picked up. - container: The image to run the whole of it in, or "" for this machine. - budget: What the run may spend, or None for whatever the flow says. + flow: The flow, by the name it is offered under, a path, or a ref. + task: What it is to do. + agents: What each agent role runs; see :meth:`runner`. + envs: What each environment role is; see :meth:`runner`. + params: The flow's params, or None for its defaults. + budget: What the run may spend; only a flow humanize ships runs without one. + resume: Whether to pick up the newest run of it here, or the epic to pick up. + outworlder: Whoever is outside the run, or None for nobody. Returns: The run. Nothing has started: `run()` runs it here, `start()` on a thread. Raises: - NotAFlow: If the flow is not there, is not a flow, or takes other agents than these. + Refused: If the flow is not there, or is given what it does not declare, or is not + given what it needs -- before anything runs. """ from hmz.runtime.doing.running import Run - return Run(self.runner(flow, agents, config, resume, container, budget), task) + return Run( + self.runner( + flow, + agents=agents, + envs=envs, + params=params, + budget=budget, + resume=resume, + ), + task, + outworlder=outworlder, + ) - def exec(self, argv: list[str]) -> None: - """Runs the flow one `hmz exec` line names, on the agents it names, to its return. + def exec(self, argv: list[str]) -> Any: + """Runs the flow one `hmz exec` line names, on what it names, to its return. Args: argv: The line, as `hmz exec` takes it. + Returns: + What the flow returned. + Raises: - NotAFlow: If the line names a flow that is not there, or takes other agents than it - declares -- which is a line that was wrong before anything ran. + Refused: If the line names a flow that is not there, or gives it what it does not + take -- which is a line that was wrong before anything ran. SystemExit: If the line is not one argparse accepts. """ - # What the line said about who is reading is the command line's to act on: this - # answers with the run itself rather than with a rendering of it. - flow, agents, task, config, budget, _ = self.read(argv) + line = self.read(argv) # Through a run, which is the one thing a flow being driven is: whoever ran a line # through this and whoever built a run are then holding the same thing. - self.run(flow, agents, task, config, budget=budget).run() + return self.run( + line.flow, + line.task, + agents=line.agents, + envs=line.envs, + params=line.params, + budget=line.budget, + resume=line.resume, + ).run() diff --git a/src/hmz/runtime/doing/epics.py b/src/hmz/runtime/doing/epics.py index 1d939ff3..4b6efdb1 100644 --- a/src/hmz/runtime/doing/epics.py +++ b/src/hmz/runtime/doing/epics.py @@ -1,7 +1,7 @@ """The runs of a workspace that have already happened, and what is gathered out of them. -One run is one epic: a directory holding what happened, what each session was logged to, and what a -flow that says it can be picked up left behind. What is written down as a run happens is +One run is one epic: a directory holding what happened, what each session was logged to, and the +journal of a flow that says it can be picked up. What is written down as a run happens is :mod:`hmz.runtime.epic`; reading the backends' own logs back is :mod:`hmz.runtime.tracing`; packaging one whole run up to send somewhere is :mod:`hmz.runtime.exporting`. All three are asked here, so that whatever is listing the runs -- a command line, the interface's own `/epics` -- asks @@ -68,13 +68,23 @@ def opened(self, epic: Path) -> dict[str, list[str]]: return opened(epic) def resumed(self, flow: str) -> Path | None: - """The last run of one flow here, which is what running a resumable flow picks up.""" + """The newest run of one flow here that can be picked up, which `--resume` picks up. + + Args: + flow: The flow, by its canonical ref or as it was named when it was run. + """ from hmz.runtime.epic import resumed return resumed(flow, self._workspace) + def picks_up(self, epic: Path) -> bool: + """Whether a run can be picked up from one epic: it kept a journal, and wrote in it.""" + from hmz.runtime.epic import picks_up + + return picks_up(epic) + def state(self, epic: Path, flow: str = "") -> dict[str, Any]: - """What a flow that says it can be picked up left behind in one run.""" + """What a resumable flow kept in one run: the run's own flow, or one it called by ref.""" from hmz.runtime.epic import state return state(epic, flow) diff --git a/src/hmz/runtime/doing/flows.py b/src/hmz/runtime/doing/flows.py index ee29f930..6b917e8a 100644 --- a/src/hmz/runtime/doing/flows.py +++ b/src/hmz/runtime/doing/flows.py @@ -1,6 +1,6 @@ """The flows there are, and the places they come from, as two objects rather than two modules. -What a flow is, is :mod:`hmz._legacy_flows`; finding one, reading one and driving one is +What a flow is, is :mod:`hmz.flows`; finding one, reading one and running one is :mod:`hmz.runtime.flowing`, and where the fetched ones are kept is the `verses` inside it. All of it is reached from here so that a command line, an interface and a daemon ask the one object rather than three modules apiece -- and so that the handful of @@ -15,12 +15,8 @@ if TYPE_CHECKING: import os from pathlib import Path - from typing import Any - from pydantic import BaseModel - - from hmz.coganchor.agents.allowance import Allowance - from hmz.runtime.flowing import Finding, Flowverse, Offer, Place, Prophecy, Running + from hmz.runtime.flowing import Declaration, Flowverse, LiveCall, Offer __all__ = ["Flows", "Flowverses"] @@ -241,143 +237,29 @@ def about(self, named: str) -> str: return about(named) - def places(self, named: str | os.PathLike[str]) -> tuple[Place, ...]: - """Every agent a flow needs chosen for it, in the order it takes them.""" - from hmz.runtime.flowing import wanted - - return wanted(named) - - def check( - self, named: str | os.PathLike[str], *, static: bool = False - ) -> tuple[Finding, ...]: - """Reads a flow for what will not run, before anything runs it. - - Two readings, in their order. The static one is pure `ast` over every file the - flow holds and executes nothing, which is the whole of what `static` keeps. The - second loads the flow and reads its live config model, in a subprocess held to a - clock -- and is left out where the first found an error: a flow that cannot run is - not one to run to find out more about. - - Args: - named: The flow, by the name `-f` takes or by a path. - static: Only the reading that executes nothing. - - An atlas gets the stricter of the two static readings, which is the compiling: its - body is a declaration rather than a program. It is chosen here rather than deeper - down because this is where both halves of the name are held, and which of the - atlases a file holds was asked for is half of it. - - Returns: - Every finding, the static reading's first and nothing said twice: a finding the - static reading already made is not repeated off the live model. - """ - from hmz.runtime.flowing import ( - checked, - inside, - is_atlas, - prophesied, - proved, - reading, - ) - - whole = reading(str(named)) - found = list( - prophesied(whole, name=inside(str(named))).findings - if is_atlas(whole) - else checked(whole) - ) - if static or any(one.severity == "error" for one in found): - return tuple(found) - # By what each said and not by its code alone: the two readings make the same - # findings about different fields, and one dropped for sharing a code with another - # is a field nothing ever mentions. - proof = proved(whole, name=inside(str(named)), scenarios=()) - said = {(one.code, one.said) for one in found} - found.extend(one for one in proof.findings if (one.code, one.said) not in said) - return tuple(found) - - def prophecy(self, named: str | os.PathLike[str]) -> Prophecy | None: - """What an atlas compiles to, read without running any of it. - - An atlas is a flow whose body is a graph: it is checked and compiled before - anything runs, and what runs is the prophecy that compiling made. This is that - prophecy -- the nodes, the edges, the shapes that flow along them, and one of these - again for every supernode. - - Args: - named: The flow, by the name `-f` takes or by a path. - - Returns: - The prophecy, or None for a flow that is not an atlas or does not compile -- - which :meth:`check` says the reasons for. - """ - from hmz.runtime.flowing import inside, prophesied, reading - - return prophesied(reading(str(named)), name=inside(str(named))).prophecy + def declared(self, named: str | os.PathLike[str]) -> Declaration: + """Everything a flow declares: its agent and environment roles, params and marks. - def foretell(self, named: str | os.PathLike[str]) -> str: - """Compiles an atlas and writes the prophecy into its own directory. - - What lands is `prophecy.pkl`, which every run of that flow from then on walks - instead of compiling the atlas again: a repository that has been through the - compiling once has an answer worth shipping. :meth:`check` says when the file and - the source it came from have drifted apart. + What a picker offers a flow's roles from, and what a line naming them is read + against -- the roles the runtime fills among them, marked as such. Args: - named: The flow, by the name `-f` takes or by a path. + named: The flow, by the name `-f` takes, a path, or a ref. Returns: - Where it was written. + The declaration. Raises: - NotAFlow: If it is not an atlas, does not compile, or is a flow that is a single - file -- which has no directory of its own to ship anything in, what is beside - such a flow being the other flows. - """ - from pathlib import Path - - from hmz.runtime.flowing import PROPHECY, NotAFlow, at, kept - - held = self.prophecy(named) - if held is None: - raise NotAFlow( - f"{named}: not an atlas that compiles -- " - f"Hmz().flows.check({named!r}) says why" - ) - # "" for a flow that is a single file, which has no directory of its own: what is - # beside such a flow is the other flows, and none of it came with this one. - beside = at(str(named)) - if not beside: - raise NotAFlow( - f"{named}: a flow that is one file has no directory to ship a prophecy " - "in -- make it a directory with an __init__.py in it" - ) - into = Path(beside) / PROPHECY - into.write_bytes(kept(held)) - return str(into) - - def configures(self, named: str | os.PathLike[str]) -> type[BaseModel] | None: - """What a flow can be set up with, or None for one that takes no setting up.""" - from hmz.runtime.flowing import configures - - return configures(named) - - def declared(self, named: str | os.PathLike[str]) -> Allowance | None: - """What a flow says a run of it may spend by default, or None for one with no opinion. - - `Allowance()` is neither: it is a flow saying in its own file that it is meant to run - under nothing at all, which is what keeps a conversation from being asked to confirm - an unbounded run every time it is picked. + FlowException: If the flow cannot be loaded -- not there, not a ref, written wrong + -- as the flow API names what went wrong. """ - from hmz.runtime.flowing.driving import declared + from hmz.runtime.flowing import resolved - return declared(named) + return resolved(str(named)).describe() def resumes(self, named: str | os.PathLike[str]) -> bool: """Whether a flow says it can be picked up where the last run of it left off.""" - from hmz.runtime.flowing import resumes - - return resumes(named) + return self.declared(named).resumable def fork(self, named: str, into: str | os.PathLike[str] | None = None) -> str: """Copies a flow into this project's own flows, whole -- what it imports and all. @@ -402,36 +284,12 @@ def fork(self, named: str, into: str | os.PathLike[str] | None = None) -> str: return fork(named, into) - def running(self) -> tuple[Running, ...]: - """Every flow running in this process now: the branch inside one, all of them outside. + def running(self) -> tuple[LiveCall, ...]: + """Every flow call going in this process now, oldest first. - Asked from inside a flow it answers with the branch that flow is on -- the one - somebody started, then each flow called to get there. Asked from anywhere else it - answers with every flow of the run, oldest first, each saying how deep it is and - what called it. + Each says its flow, how deep it is and which call made it: the running tree, read the + same way from anywhere, inside a flow or out. """ from hmz.runtime.flowing import running return running() - - def set_up_from( - self, said: str | os.PathLike[str] - ) -> tuple[dict[str, Any] | None, Allowance | None]: - """Reads what a flow is to be set up with, and what a run of it may spend, out of YAML. - - Args: - said: The path to the file. - - Returns: - What it holds field by field with the run's own `budget:` taken out of it -- that - one being a setting of the run rather than of the flow -- or None where it left - nothing for the flow at all. And the allowance that key said, or None where the - file said nothing about one. - - Raises: - ValueError: If the file cannot be read, holds something that is not a mapping, or - says a budget that cannot be read as one. - """ - from hmz.runtime.runner import set_up_from - - return set_up_from(said) diff --git a/src/hmz/runtime/doing/running.py b/src/hmz/runtime/doing/running.py index 96b058cc..bc1ce8eb 100644 --- a/src/hmz/runtime/doing/running.py +++ b/src/hmz/runtime/doing/running.py @@ -1,68 +1,133 @@ """A flow that is running, and the handful of things there are to do to one. -:class:`hmz.runtime.runner.Runner` is a flow loaded and handed its agents; running it is a call that -returns when the flow does, which for a loop meant to run for a week is not a call anything -holding a terminal can make. This is that call put on a thread of its own, with the two things -somebody watching a run asks for -- whether it is still going, and to stop it. +:class:`hmz.runtime.runner.Runner` is a flow loaded and handed its drivers; running it is a +coroutine that returns when the flow does, which for a loop meant to run for a week is not a +call anything holding a terminal can make. This is that coroutine put on a loop of its own -- +here, or on a thread of its own -- with the things somebody watching a run asks for: what it +has opened, what it has spent, whether it is still going, and to stop it. """ from __future__ import annotations -from typing import TYPE_CHECKING +import asyncio +import threading +from typing import TYPE_CHECKING, Any if TYPE_CHECKING: - import threading - - from hmz.coganchor.agents import AgentBase + from collections.abc import Callable + from pathlib import Path + + from hmz.coganchor.agents import AgentBase, SessionBase + from hmz.flows import Budget, Usage + from hmz.runtime.epic import Epic + from hmz.runtime.flowing import Declaration, OutworlderDriver + from hmz.runtime.flowing.harnesses import Listener from hmz.runtime.runner import Runner __all__ = ["Run"] class Run: - """One run of one flow: the agents driving it, and how it ends.""" + """One run of one flow: what it opened, what it spent, and how it ends.""" - def __init__(self, runner: Runner, task: str) -> None: - """Holds a loaded flow and what it is to have its agents do. + def __init__( + self, runner: Runner, task: str, *, outworlder: OutworlderDriver | None = None + ) -> None: + """Holds a loaded flow and what it is to do. Nothing is started here: a run is started by :meth:`start`, or run to its return by :meth:`run`, so that whoever made one chooses which of the two they are holding. Args: - runner: The flow, loaded and handed the agents it declared. - task: What the flow is to have them do. + runner: The flow, loaded and handed its drivers. + task: What it is to do. + outworlder: Whoever is outside the run, or None for nobody -- an outworlder that is + always away, which is what a command line is. """ self._runner = runner self._task = task + self._outworlder = outworlder + self._opened: list[Callable[[str, AgentBase, SessionBase], None]] = [] + self._agents: list[AgentBase] = [] + self._epic: Path | None = None self._thread: threading.Thread | None = None + self._loop: asyncio.AbstractEventLoop | None = None + self._running: asyncio.Task[Any] | None = None + self._stopping = False self._raised: BaseException | None = None + self._result: Any = None + self._lock = threading.Lock() + + # ----------------------------------------------------------------------- what it is + + @property + def flow(self) -> str: + """The flow, as it was named.""" + return self._runner.flow + + @property + def ref(self) -> str: + """The flow's canonical ref, which is what the running tree names it by.""" + return self._runner.impl.ref + + @property + def task(self) -> str: + """What it was asked to do.""" + return self._task + + @property + def declaration(self) -> Declaration: + """What the flow declares.""" + return self._runner.declaration + + @property + def budget(self) -> Budget: + """What the run may spend.""" + return self._runner.budget + + @property + def usage(self) -> Usage: + """What every session of the run has spent so far.""" + from hmz.flows import Usage + + recorder = self._runner.recorder + return Usage() if recorder is None else recorder.usage() @property def agents(self) -> tuple[AgentBase, ...]: - """Every agent this drives, the person the flow talks to among them.""" - return self._runner.agents + """The coganchor agent behind each session the run has opened, oldest first. + + Each is named for the role it was opened for, which is what its events say. + """ + with self._lock: + return tuple(self._agents) @property - def unwatched(self) -> bool: - """Whether nothing at all will stop this run and nobody has said that is the point.""" - return self._runner.unwatched + def epic(self) -> Path | None: + """The epic the run is written into, once it has started.""" + return self._epic def unreadable(self) -> str: - """Which of the caps this run was given nothing in it can read, in words. + """Which cap of the run nothing it drives can read, in words, or "" for none.""" + return self._runner.unreadable() - Returns: - One line about them, or "" where every cap set can be read. + def watch(self, listener: Listener) -> None: + """Has everything every session of the run says reach `listener`, as it is said. + + Args: + listener: What to tell -- the agent, the conversation and the event -- from + whichever thread a CLI is read on. """ - return self._runner.unreadable() + self._runner.watch(listener) - def unserved(self) -> str: - """Which of this flow's declarations its agents' backends could not carry, in words. + def opened(self, callback: Callable[[str, AgentBase, SessionBase], None]) -> None: + """Has each session the run opens told to `callback` as it opens. - Returns: - One line per declaration given up by a place that wrote `insist=False`, or "" for - a run that carried everything its flow declared. + Args: + callback: What to tell: the role, the coganchor agent and its conversation. Told + on the run's own loop, before the session's first turn. """ - return self._runner.unserved() + self._opened.append(callback) @property def running(self) -> bool: @@ -74,30 +139,68 @@ def raised(self) -> BaseException | None: """Whatever the flow raised, for a run started on a thread of its own and now over.""" return self._raised - def run(self) -> None: + @property + def result(self) -> Any: + """What the flow returned, for a run started on a thread of its own and now over.""" + return self._result + + # ------------------------------------------------------------------------- running + + async def _main(self) -> Any: + """The run, on whichever loop is running it.""" + with self._lock: + self._loop = asyncio.get_running_loop() + self._running = asyncio.current_task() + stopping = self._stopping + if stopping: + # Stopped before it began: nothing ran, and what it was given goes all the same. + await self._runner.aclose() + raise asyncio.CancelledError + return await self._runner.arun( + self._task, + outworlder=self._outworlder, + opened=self._told, + started=self._began, + ) + + def _told(self, role: str, agent: AgentBase, session: SessionBase) -> None: + with self._lock: + self._agents.append(agent) + for callback in tuple(self._opened): + callback(role, agent, session) + + def _began(self, epic: Epic) -> None: + self._epic = epic.path + + def run(self) -> Any: """Runs the flow here, until it returns. + On a loop of its own in this thread, which is where a signal reaches it; from a + thread already running a loop -- an interface, a test -- on a thread of its own, the + flow's turns being waited on by a loop this one must not hold. + + Returns: + What the flow returned. + Raises: BaseException: Whatever the flow raised, as it raised it. """ - self._runner.run(self._task) + try: + asyncio.get_running_loop() + except RuntimeError: + return asyncio.run(self._main()) + self.start() + self.wait() + if self._raised is not None: + raise self._raised + return self._result def start(self) -> None: """Starts the flow on a thread of its own, and returns at once. - One run in a container at a time, per process: the container a run works in is the - process's, since a flow that called another is one run working in one place -- so two - of these started at once with an image between them would be two runs reaching for - one container. Runs on this machine have no such thing between them. The second of - two is refused where the container is settled, which is on the thread this starts -- - so what says so is :attr:`raised` rather than this call, and a caller holding two - runs in containers has to read it. :meth:`run` raises it where it stands. - Raises: RuntimeError: If it has already been started. """ - import threading - if self._thread is not None: raise RuntimeError("this run has already been started") self._thread = threading.Thread( @@ -106,9 +209,9 @@ def start(self) -> None: self._thread.start() def _drives(self) -> None: - """Runs the flow, keeping whatever it raised for whoever asks afterwards.""" + """Runs the flow, keeping what it returned or raised for whoever asks afterwards.""" try: - self._runner.run(self._task) + self._result = asyncio.run(self._main()) except BaseException as why: # noqa: BLE001 -- kept rather than swallowed self._raised = why @@ -127,23 +230,35 @@ def wait(self, timeout: float | None = None) -> bool: return not self._thread.is_alive() def stop(self) -> None: - """Tells every agent to take no further turn, so the loop ends rather than handing on. + """Stops the flow: the turn under way is interrupted, and the flow unwinds. - The turn running now is closed out first: a flow told to stop unwinds in its own time. - :meth:`close` is what does not wait for it. + Every call of it raises where it stands, every session it opened is closed and every + temporary directory it made is taken away -- in its own time, which :meth:`close` does + not wait for. From any thread. """ - for agent in self.agents: - agent.stop() + with self._lock: + self._stopping = True + loop, running = self._loop, self._running + if loop is None or running is None or loop.is_closed(): + return + try: + loop.call_soon_threadsafe(running.cancel) + except RuntimeError: # the loop closed between the two + return def close(self) -> None: - """Closes every conversation still open, which is the backend's process going. + """Stops the flow and ends every conversation still open, without waiting for it. What the flow gets back is a turn that failed, the same thing it would have got had the agent fallen over by itself. The last thing there is to do about a run. """ import contextlib + self.stop() + recorder = self._runner.recorder + for handle in () if recorder is None else recorder.sessions: + with contextlib.suppress(Exception): + handle.interrupt() for agent in self.agents: - for session in agent.sessions: - with contextlib.suppress(Exception): - session.close() + with contextlib.suppress(Exception): + agent.stop() diff --git a/src/hmz/runtime/epic.py b/src/hmz/runtime/epic.py index 950f9b88..679eaca4 100644 --- a/src/hmz/runtime/epic.py +++ b/src/hmz/runtime/epic.py @@ -19,14 +19,14 @@ ~/.humanize/epics//-/ epic.jsonl what happened, a line at a time epic._.jsonl the same, for one flow the run called - state.json what a flow that can be picked up again left behind + resume.jsonl the engine's journal, for a flow that can be picked up profile.jsonl the programs it ran, for a run that was profiled sessions//… a link per file the backend logged it to traces/.trace.json what was gathered of it afterwards, to be read -A flow may call another, and a called flow opens sessions and keeps state exactly as the flow -that called it does. So each call gets a record of its own beside the run's own, and the -record of whatever called it says what it called and which file to read it in. Still one run +A flow may call another, and a called flow opens sessions exactly as the flow that called it +does. So each call gets a record of its own beside the run's own, and the record of whatever +called it says what it called and which file to read it in. Still one run and still one directory: a called flow is part of the run that called it, not another run. One directory and one tree. A call made from inside a called flow is written under *that* @@ -37,16 +37,20 @@ It opens when the flow starts and closes when the flow stops, however it stops -- finished, failed, or interrupted. A closed epic is never reopened: running the flow again is another run, with sessions of its own, and so another epic -- which is what a flow that says it can -be picked up again is picked up as. What it left behind is read out of the epic it left it -in and handed to the next run of it, which writes into an epic of its own. +be picked up again is picked up as. What a resumable flow keeps is the engine's journal +(:mod:`hmz.runtime.flowing.journaling`), written into the epic of the run keeping it; a run +picking it up is handed a copy of it in an epic of its own, which the engine compacts and goes +on appending to. """ from __future__ import annotations +import asyncio import contextlib import datetime import json import re +import shutil import threading import uuid from collections import Counter @@ -57,7 +61,7 @@ from hmz.coganchor import backends if TYPE_CHECKING: - from collections.abc import Mapping, Sequence + from collections.abc import Iterator, Mapping, Sequence from hmz.coganchor.agents import AgentBase @@ -68,20 +72,20 @@ "LOCAL", "RECORD", "RECORDS", + "RESUME", "SESSIONS", - "STATE", "TRACES", "Called", "Drove", "Epic", "Ran", "Session", - "State", "Sub", "called", "epics", "linked", "opened", + "picks_up", "read", "records", "resumed", @@ -116,8 +120,9 @@ #: Where the links to the sessions' own logs go, a directory per session. SESSIONS = "sessions" -#: What a resumable flow left behind, kept beside the run it left it in. -STATE = "state.json" +#: The engine's journal of a resumable run: what each flow call kept, and which calls ended. +#: What `--resume` and `/resume` pick a run up from, kept beside the run it was written by. +RESUME = "resume.jsonl" #: Where the traces gathered of one run go, inside that run's own directory. A trace of a run #: belongs with the run: the sessions it points at and the state it left are already there. @@ -177,36 +182,27 @@ class Session(NamedTuple): class Drove(NamedTuple): - """One agent a run was driven by, as the run wrote it down. + """One agent a run was given, as the run wrote it down. Attributes: - agent: What the flow calls it. + agent: The role the flow calls it by. backend: The CLI it drives. model: What that CLI was asked to run. - effort: How hard it was asked to think. - permission: What it was allowed to do without being asked, or "" where the run said - nothing about it -- which is an agent allowed whatever its own CLI allows one. + effort: How hard it was asked to think, "" for the CLI's own default. provider: The account it was configured to run as, or "" for this machine's own. - goals: Whether it was allowed to run under its backend's own goal feature. - person: Whether it was the person at the prompt, who is handed to a flow rather than - chosen -- so a run picked up again is picked up on the agents somebody chose, and - the person is handed over afresh by whatever is doing the picking up. """ agent: str backend: str model: str effort: str - permission: str = "" provider: str = "" - goals: bool = True - person: bool = False @property def spec(self) -> str: - """What it runs, spelled the way `-a` spells one.""" + """What it runs, spelled the way `-a` spells one after the role.""" cli = f"{self.backend}@{self.provider}" if self.provider else self.backend - return f"{cli}/{self.model}:{self.effort}" + return f"{cli}/{self.model}:{self.effort or 'auto'}" class Called(NamedTuple): @@ -248,7 +244,7 @@ class Ran(NamedTuple): began: When it started. ended: When it stopped, or "" for one still running or abandoned where it stood. how: How it stopped -- done, failed or stopped -- and "" while it has not. - agents: What drove it, in the order the flow takes them. + agents: What it was given for each agent role, in the order the flow declares them. sessions: Every session it opened, oldest first, the ones opened inside a flow it called among them -- one run is one run, however many flows it took to run it. called: Every flow this run called, in the order it called them. What each of those @@ -256,6 +252,12 @@ class Ran(NamedTuple): resumable: Whether the flow said it could be picked up again when this run happened. Whether it says so now is asked of the flow: this is what the run recorded, which is what it was rather than what can be done with it today. + ref: The flow's canonical ref, which is what a run is picked up by whatever it was + named as; "" for a run written before there was one. + envs: What it was given for each environment role, each as `-e` spells one. + params: What the flow was set up with, as JSON. + budget: What the run was allowed to spend, as JSON, or None where it said nothing. + picked_up: The epic this run was picked up from, by name, or "". """ at: Path @@ -269,6 +271,11 @@ class Ran(NamedTuple): sessions: tuple[Session, ...] = () called: tuple[Called, ...] = () resumable: bool = False + ref: str = "" + envs: tuple[str, ...] = () + params: dict[str, Any] = {} # noqa: RUF012 -- a NamedTuple's default, never written to + budget: dict[str, Any] | None = None + picked_up: str = "" @property def name(self) -> str: @@ -412,213 +419,135 @@ def _link(at: Path, backend: str, ident: str) -> list[str]: return made -class State(dict[str, Any]): - """What a resumable flow left behind, and what it is writing now. +def _journal(epic: Path) -> Iterator[dict[str, Any]]: + """Every record of one epic's resume journal, skipping what a killed run left half-written. - A dict as far as the flow is concerned -- it is handed one, it writes into it, and the - next run of that flow is handed what it wrote. What it also is is a file in the epic, - written as the flow writes: a flow worth picking up again is one that was stopped or - killed rather than one that ended tidily, and state saved only at the end is state a - stopped run does not have. Something written inside a value it holds -- a list appended - to, a dict of its own written into -- is a change no mapping can see, and is saved when - the run ends or when the flow says :meth:`save`. - """ - - def __init__( - self, at: Path, flow: str, held: Mapping[str, Any] | None = None - ) -> None: - """Holds what one flow left behind, against the epic it is being written into. + Args: + epic: The epic's directory. - Args: - at: The epic's directory. - flow: Whose state this is, since a flow that called another is two flows and each - has its own to keep. - held: What was read back, or nothing for a run that is picking nothing up. - """ - super().__init__(held or {}) - self._at = at - self._flow = flow - self._writing = threading.Lock() - - def __setitem__(self, key: str, value: Any) -> None: - super().__setitem__(key, value) - self.save() - - def __delitem__(self, key: str) -> None: - super().__delitem__(key) - self.save() - - def update(self, *said: Any, **and_so: Any) -> None: - super().update(*said, **and_so) - self.save() - - def setdefault(self, key: str, default: Any = None) -> Any: - held = super().setdefault(key, default) - self.save() - return held - - def pop(self, *said: Any) -> Any: - held = super().pop(*said) - self.save() - return held - - def popitem(self) -> tuple[str, Any]: - held = super().popitem() - self.save() - return held - - def clear(self) -> None: - super().clear() - self.save() - - def save(self) -> None: - """Writes what this flow is holding into the epic, beside what the others hold. - - Read again and merged rather than dumped over, for the reason the settings are: a - flow that called another is two flows writing one file, and a plain dump would put - back a file missing whatever the other had written. Whole and then moved into place, - so that one read while it is being written is the old one or the new one. - - Anything that cannot be written -- a value no JSON has a shape for, a directory that - has gone -- leaves the run as it was: state is what a flow may pick up, and a run - that stopped because it could not save it would be worse than one that cannot. - """ - with self._writing: - held = _kept(self._at) - held[self._flow] = dict(self) - try: - self._at.mkdir(parents=True, exist_ok=True) - said = json.dumps(held, ensure_ascii=False, default=str) - beside = self._at / f".{STATE}.new" - beside.write_text(said, encoding="utf-8") - beside.replace(self._at / STATE) - except (OSError, TypeError, ValueError): - return + Yields: + One record apiece, and nothing at all for an epic that keeps no journal. + """ + try: + lines = (epic / RESUME).read_bytes().splitlines() + except OSError: + return + for line in lines: + try: + said = json.loads(line) + except ValueError: + continue + if isinstance(said, dict): + yield cast("dict[str, Any]", said) -def _kept(epic: Path) -> dict[str, Any]: - """What every flow of one epic left behind, by the name each was run as. +def picks_up(epic: Path) -> bool: + """Whether a run can be picked up from one epic: whether its journal holds a flow call. Args: epic: The epic's directory. Returns: - One entry per flow that wrote anything, and nothing at all for an epic that holds no - state, holds one nothing can read, or holds one written by hand as something else. + True where the run was of a resumable flow and got as far as writing its first call + down; False for any other run, and for one killed before it wrote anything. """ - try: - said = json.loads((epic / STATE).read_text(encoding="utf-8")) - except (OSError, ValueError): - return {} - if not isinstance(said, dict): - return {} - return { - str(flow): cast("dict[str, Any]", one) - for flow, one in cast("dict[str, Any]", said).items() - if isinstance(one, dict) - } + return any(one.get("t") == "call" for one in _journal(epic)) def state(epic: Path, flow: str = "") -> dict[str, Any]: - """What a resumable flow left behind in one epic. + """What a resumable flow kept in one run, as the run left it. Args: epic: The epic's directory. - flow: Which flow's, as it was named when it ran, or "" for the one the epic is a run - of -- which is the flow somebody picking the epic up is picking up. + flow: Which flow's, by its canonical ref -- the last call of it the run made -- or "" + for the flow the run was of. Returns: - What it wrote, and nothing at all where that flow wrote nothing. + What it kept, key by key, and nothing at all where that flow kept nothing or the run + kept no journal. """ - held = _kept(epic) - if flow: - return held.get(flow, {}) - ran = read(epic) - return held.get(ran.flow, {}) if ran is not None else {} + calls: dict[int, str] = {} + held: dict[int, dict[str, Any]] = {} + which: int | None = None + for said in _journal(epic): + kind = said.get("t") + try: + ident = int(said["id"]) + except (KeyError, TypeError, ValueError): + continue + if kind == "call": + calls[ident] = str(said.get("ref") or "") + held[ident] = {} + if (not flow and said.get("parent") == 0 and which is None) or ( + flow and calls[ident] == flow + ): + which = ident + elif kind == "set" and ident in held: + held[ident][str(said.get("key"))] = said.get("value") + elif kind == "del" and ident in held: + held[ident].pop(str(said.get("key")), None) + return held.get(which, {}) if which is not None else {} def resumed(flow: str, workspace: Path | str | None = None) -> Path | None: - """The epic one flow's next run picks up from, which is the last run of it here. + """The epic a resumable flow's next run picks up from: the newest one it can be. Args: - flow: The flow, as it is named when it is run. + flow: The flow, by its canonical ref or as it was named when it was run. workspace: Where it runs, defaulting to this directory. Returns: - The epic, or None where the flow has not run here. A run that wrote nothing at all is - nothing to pick up and the search goes past it; a run that wrote and then emptied what - it had written is not the same thing, and is where the search stops -- a flow that - cleared its state said the next run starts clean, and answering that by handing it the - state of the run before would be answering the opposite. Found by what the state holds - rather than by what the run was of, so that a flow which was called by another is - picked up too -- it wrote under its own name, which is where it is looked for. + The newest epic of that flow here whose run was resumable and wrote its journal, or + None where there is none -- a flow that never ran here, ran as something that could + not be picked up, or was killed before it wrote anything down. """ for epic in reversed(epics(workspace)): - if flow in _kept(epic): + began = next( + (one for one in _events(epic) if one.get("event") == "began"), None + ) + if began is None or not began.get("resumable"): + continue + if flow not in (began.get("ref"), began.get("flow")): + continue + if picks_up(epic): return epic return None -def _drove(agents: Sequence[AgentBase]) -> list[dict[str, Any]]: - """What each agent of a run is, for the line a record opens with. - - Args: - agents: The agents, in the order the flow takes them. - - Returns: - One entry apiece, saying what it drives and at what. - """ - from hmz.coganchor.agents import HumanAgent - - return [ - { - "agent": agent.id, - "backend": agent.backend, - "model": agent.config.model, - "effort": agent.config.effort, - "service_tier": agent.config.service_tier, - "permission": agent.config.permission, - # What it was configured with rather than what a turn of it ends up running as: - # the account a turn fell back onto is written down against the session that ran - # there, which is where it happened. - "provider": agent.config.provider, - "goals": agent.config.goals, - "web_search": agent.config.web_search, - # Asked as the run is written down rather than read back off a name: what the - # person's backend is called is the agents' own business, and what a run picked - # up again needs is which of its agents nobody chose. - "person": isinstance(agent, HumanAgent), - } - for agent in agents - ] - - class Epic: """One run of one flow: the directory it is written to, and what has happened to it.""" def __init__( self, flow: str, - agents: Sequence[AgentBase], task: str, workspace: Path | None = None, *, + ref: str = "", + agents: Sequence[Drove] = (), + envs: Sequence[str] = (), + params: Mapping[str, Any] | None = None, + budget: Mapping[str, Any] | None = None, resumable: bool = False, - picked_up: str = "", + picked_up: Path | None = None, profile: bool = False, ) -> None: """Opens an epic, and writes down what it is a run of. Args: flow: The flow being run, as it was named. - agents: The agents it is being run with, in the order it takes them. - task: What they were asked to do. + task: What its agents were asked to do. workspace: Where the run happens, defaulting to this directory. Epics are kept under the workspace they ran in, since that is what anyone looking for one has. + ref: The flow's canonical ref, which is what a run is picked up by. + agents: What each agent role was given, in the order the flow declares them. + envs: What each environment role was given, as `-e` spells one. + params: What the flow was set up with, as JSON. + budget: What the run may spend, as JSON, or None. resumable: Whether the flow says it can be picked up again, which is what makes - the state it leaves behind something to run it on rather than something to read. - picked_up: The epic this run was picked up from, by name, or "" for one starting - from nothing. + the journal it keeps something to run it on rather than something to read. + picked_up: The epic this run is picked up from, whose journal is copied into this + one for the run to go on from, or None for a run starting from nothing. profile: Whether to sample the programs the agents start while the run goes, so that what a turn spent its minutes on is in the run's trace beside the turn. A setting of the workspace, asked of it by whoever opens the epic. @@ -636,8 +565,13 @@ def __init__( JOURNAL, (workspace or Path.cwd()).resolve(), flow, - agents, ) + if picked_up is not None: + # A copy rather than the file itself: the run it came from is closed, and the + # engine compacts what it picks up before it appends to it. + self._at.mkdir(parents=True, exist_ok=True) + with contextlib.suppress(OSError): + shutil.copyfile(picked_up / RESUME, self._at / RESUME) #: The programs this run starts, sampled while it runs, or None for a run nobody #: asked to profile -- which is every run until somebody says otherwise. self._profiler = self._profiling() if profile else None @@ -647,46 +581,41 @@ def __init__( task=task, workspace=str(self._where), resumable=resumable, - **({"picked_up": picked_up} if picked_up else {}), - agents=_drove(agents), + **({"ref": ref} if ref else {}), + **({"picked_up": picked_up.name} if picked_up is not None else {}), + agents=[one._asdict() for one in agents], + envs=list(envs), + params=dict(params or {}), + **({"budget": dict(budget)} if budget is not None else {}), ) - def _begin( - self, - at: Path, - journal: str, - workspace: Path, - flow: str, - agents: Sequence[AgentBase], - ) -> None: + def _begin(self, at: Path, journal: str, workspace: Path, flow: str) -> None: """Settles what is written down, and where. Shared with the record of a flow this one called, which is the same thing written - into a file of its own beside this one: a called flow opens sessions and keeps state - exactly as the flow that called it does, and neither writes the other's. + into a file of its own beside this one: a called flow opens sessions exactly as the + flow that called it does, and neither writes the other's. Args: at: The epic's directory. journal: The file inside it these lines go to. workspace: Where the run is happening. flow: The flow this is a record of, as it was named. - agents: The agents it is being run with, in the order it takes them. """ self._at = at self._journal = journal self._writing = ( threading.Lock() ) # sessions open on whichever thread a turn runs on - self._agents = list(agents) #: Every session this run has opened, by the name it was written down under, so that #: the links can be made again as the backends go on writing to them. self._sessions: dict[str, tuple[str, str]] = {} - #: What each resumable flow of this run is holding, so that a value written inside - #: one -- which no mapping can see -- is still saved when the run ends. - self._state: list[State] = [] self._flow = flow self._where = workspace self._profiler: Profiler | None = None + #: How it ended, where whoever is running it has said: "stopped" for a run stopped + #: by hand or by what it was allowed to spend, rather than one that failed. + self._how = "" @property def path(self) -> Path: @@ -708,6 +637,20 @@ def workspace(self) -> Path: """Where this run is happening, which is what its epics are kept under.""" return self._where + @property + def resume(self) -> Path: + """Where the engine keeps this run's journal, for a flow that can be picked up.""" + return self._at / RESUME + + def stopped(self) -> None: + """Says the run was stopped rather than failed, for the line it ends with. + + A run stopped by hand, or by the budget it was given, is the ordinary end of a run + nothing else ends; what raised out of it is still what stopped it, and is not a + failure to report. + """ + self._how = "stopped" + def _profiling(self) -> Profiler | None: """The sampler this run is profiled by, started, or None where there is none. @@ -730,22 +673,6 @@ def _profiling(self) -> Profiler | None: return None return one - def state(self, flow: str = "", held: Mapping[str, Any] | None = None) -> State: - """The dict a resumable flow of this run writes what it wants back into. - - Args: - flow: Whose it is, as that flow was named, or "" for the flow this is a run of. - A flow that called another is two flows, and each keeps its own. - held: What it is picking up, or nothing for a run starting from nothing. - - Returns: - The state, saved into this epic as the flow writes it. - """ - one = State(self._at, flow or self._flow, held) - with self._writing: - self._state.append(one) - return one - def __enter__(self) -> Self: """Hands the epic to whatever is running the flow inside it.""" return self @@ -761,8 +688,6 @@ def __exit__( traceback: Where it was raised, unread. """ self._close(kind) - for agent in self._agents: - agent.epic = None def _close(self, kind: type[BaseException] | None) -> None: """Writes down that what this is a record of has ended, and how it ended. @@ -773,8 +698,6 @@ def _close(self, kind: type[BaseException] | None) -> None: Args: kind: What was raised out of it, if anything. """ - from hmz.coganchor.agents import Stopped - # The sampler first, so that what it saw is written down before anything reads it, # and so that a run which is over stops costing anything. if self._profiler is not None: @@ -783,41 +706,30 @@ def _close(self, kind: type[BaseException] | None) -> None: # the session runs and finishes writing it after the last turn, and a sub-agent's # transcript appears whenever that sub-agent was started. self.links() - # And what each flow of this run is holding, which is where a value written inside - # something the state holds -- a list appended to -- is finally written down. - for one in list(self._state): - one.save() - # An agent that was told to stop is a run that was stopped, whatever the turn under - # way made of it: the process goes out from under that turn, and from inside one that - # reads as a turn that could not finish. - stopped = kind is not None and ( - issubclass(kind, Stopped) or any(agent.stopped for agent in self._agents) + # A run interrupted from outside is a run that was stopped, however the turn under + # way made of it: the process goes out from under that turn, and from inside one + # that reads as a turn that could not finish. + stopped = kind is not None and issubclass( + kind, KeyboardInterrupt | asyncio.CancelledError ) self.write( "ended", - how="stopped" if stopped else "failed" if kind is not None else "done", + how=self._how + or ("stopped" if stopped else "failed" if kind is not None else "done"), ) - def called( - self, - flow: str, - agents: Sequence[AgentBase], - task: str, - *, - resumable: bool = False, - ) -> Sub: + def called(self, flow: str, task: str = "", *, resumable: bool = False) -> Sub: """Opens the record of a flow this one called, beside this one's own. - A flow that called another is two flows, and each of them opened sessions, kept its - own state and may have called a third. So each gets a record of its own -- one file - per call, in the directory of the run that started it -- and this one is left saying - what it called, when, and which file to read it in. One run, written down as the - shape it actually ran in rather than as one flat list nothing can be attributed to. + A flow that called another is two flows, and each of them opened sessions and may + have called a third. So each gets a record of its own -- one file per call, in the + directory of the run that started it -- and this one is left saying what it called, + when, and which file to read it in. One run, written down as the shape it actually + ran in rather than as one flat list nothing can be attributed to. Args: - flow: The flow being called, as it was asked for. - agents: The agents it was handed, in the order it takes them. - task: What it was called with. + flow: The flow being called, by its canonical ref. + task: What it was called with, where that is known. resumable: Whether it says it can be picked up again. Returns: @@ -827,7 +739,7 @@ def called( # runs of it, each with its own sessions, and one file for both would say neither. record = _record(flow, uuid.uuid4().hex[:6]) self.write("called", flow=flow, task=task, epic=record) - return Sub(self, record, flow, agents, task, resumable=resumable) + return Sub(self, record, flow, task, resumable=resumable) def opened(self, agent: AgentBase, session: str, parent: str = "") -> None: """Writes down a session one of the agents has just opened. @@ -844,16 +756,29 @@ def opened(self, agent: AgentBase, session: str, parent: str = "") -> None: parent: The id of the conversation it was forked from, or "" for one that started from nothing. """ - provider = _provider(agent) - name = called(agent.id, agent.backend, provider, session) + self.session(agent.id, agent.backend, _provider(agent), session, parent) + + def session( + self, agent: str, backend: str, provider: str, ident: str, parent: str = "" + ) -> None: + """Writes down a session, as :meth:`opened` does, for one no coganchor agent opened. + + Args: + agent: Whose it is, by the role the flow calls that agent. + backend: What took its turns. + provider: The account they ran as, or "" for this machine's own. + ident: Its id. + parent: The id of the conversation it was forked from, or "". + """ + name = called(agent, backend, provider, ident) with self._writing: - self._sessions[name] = (agent.backend, session) + self._sessions[name] = (backend, ident) self.write( "opened", - agent=agent.id, - backend=agent.backend, + agent=agent, + backend=backend, provider=provider or LOCAL, - session=session, + session=ident, name=name, # Where to look for it inside this epic, which is a link and not the log itself. where=f"{SESSIONS}/{name}", @@ -896,7 +821,7 @@ class Sub(Epic): """One flow another flow called, written down in a record of its own. Everything a run writes down, a flow the run called writes down too: the sessions it - opened, what it kept, and whatever it called in turn. What it does not have is a + opened, and whatever it called in turn. What it does not have is a directory: it is part of the run that called it, so its record sits beside that run's own in the same epic, and its sessions link into the same `sessions/`. @@ -911,8 +836,7 @@ def __init__( under: Epic, record: str, flow: str, - agents: Sequence[AgentBase], - task: str, + task: str = "", *, resumable: bool = False, ) -> None: @@ -921,13 +845,12 @@ def __init__( Args: under: What called it, which is where the call itself is written down. record: What this record is called, inside the epic they share. - flow: The flow being called, as it was asked for. - agents: The agents it was handed, in the order it takes them. - task: What it was called with. + flow: The flow being called, by its canonical ref. + task: What it was called with, where that is known. resumable: Whether it says it can be picked up again. """ self._under = under - self._begin(under.path, record, under.workspace, flow, agents) + self._begin(under.path, record, under.workspace, flow) self.write( "began", flow=flow, @@ -937,15 +860,18 @@ def __init__( # Which record called this one, so that a flow that called a flow that called a # flow reads back as what it was rather than as three things one run did. under=under.record, - agents=_drove(agents), ) - def ended(self, kind: type[BaseException] | None = None) -> None: + def ended(self, kind: type[BaseException] | None = None, how: str = "") -> None: """Closes this record, and writes the call's other end where the call was written. Args: kind: What was raised out of the called flow, if anything. + how: How it ended where the caller knows better than `kind` says -- "stopped" + for a call stopped rather than failed -- or "" to read it off `kind`. """ + if how: + self._how = how self._close(kind) self._under.write("returned", flow=self._flow, epic=self._journal) @@ -1106,24 +1032,6 @@ def sessions(epic: Path) -> list[Session]: return sorted(held, key=lambda one: one.at) -def _allowed(said: object) -> str: - """What one agent of a run was allowed, as the record it was written in says it. - - Args: - said: What the record holds under `permission`, which is None for a record holding - nothing there at all. - - Returns: - The rung it names, "" for a run that said nothing to its agent's CLI about what it may - do, and `bypass` for a record written before a run wrote this down -- every agent of - every run then was allowed everything, and a report that showed those as the silence - would be reporting a rung nobody was ever at. - """ - from hmz.coganchor.agents import PERMISSIONS - - return str(said) if isinstance(said, str) else PERMISSIONS[-1] - - def read(epic: Path) -> Ran | None: """What one epic was, read back off its own record. @@ -1150,19 +1058,12 @@ def read(epic: Path) -> Ran | None: backend=str(said.get("backend") or ""), model=str(said.get("model") or ""), effort=str(said.get("effort") or ""), - # Two different silences, and the difference is the whole of what this - # field now says. A record with no such key at all was written before a run - # wrote one down, and every agent of every run then was allowed everything: - # it reads as `bypass`, which is what that agent actually did. A record that - # holds the key empty was written by a run that said nothing to the CLI, and - # it stays empty. Read as one silence, the older runs would report a rung - # nobody was ever at. - permission=_allowed(said.get("permission")), provider=str(said.get("provider") or ""), - goals=bool(said.get("goals", True)), - person=bool(said.get("person")), ) ) + envs = began.get("envs") + params = began.get("params") + budget = began.get("budget") return Ran( at=epic, flow=str(began.get("flow") or ""), @@ -1175,6 +1076,14 @@ def read(epic: Path) -> Ran | None: sessions=tuple(sessions(epic)), called=tuple(_calls(events)), resumable=bool(began.get("resumable")), + ref=str(began.get("ref") or ""), + envs=tuple( + str(one) + for one in cast("list[Any]", envs if isinstance(envs, list) else []) + ), + params=cast("dict[str, Any]", params) if isinstance(params, dict) else {}, + budget=cast("dict[str, Any]", budget) if isinstance(budget, dict) else None, + picked_up=str(began.get("picked_up") or ""), ) diff --git a/src/hmz/runtime/exporting.py b/src/hmz/runtime/exporting.py index 5780142c..65494202 100644 --- a/src/hmz/runtime/exporting.py +++ b/src/hmz/runtime/exporting.py @@ -50,8 +50,8 @@ from hmz.coganchor import backends from hmz.runtime.epic import ( JOURNAL, + RESUME, SESSIONS, - STATE, TRACES, read, records, @@ -389,10 +389,10 @@ def _lands(epic: Path, at: str | os.PathLike[str] | None) -> Path: def _files(epic: Path) -> Iterator[tuple[str, Path]]: """Everything one run wrote down about itself, by the name it goes in the archive under. - The run's own record first, then a record per flow it called, then what a resumable flow - left behind, the profile of the programs it ran, and every trace gathered of it. Each is - absent from a run that never wrote one -- a flow that keeps no state, a run nobody - profiled -- and an absent one is left out rather than carried as an empty file. + The run's own record first, then a record per flow it called, then the journal a + resumable flow kept, the profile of the programs it ran, and every trace gathered of it. + Each is absent from a run that never wrote one -- a flow that is not resumable, a run + nobody profiled -- and an absent one is left out rather than carried as an empty file. Args: epic: The run, by the directory it is written in. @@ -404,7 +404,7 @@ def _files(epic: Path) -> Iterator[tuple[str, Path]]: for one in records(epic): yield one.name, one - for name in (STATE, PROFILE): + for name in (RESUME, PROFILE): if (epic / name).is_file(): yield name, epic / name with contextlib.suppress(OSError): @@ -704,20 +704,14 @@ def _manifest( "backend": one.backend, "model": one.model, "effort": one.effort, - # Empty where the run said nothing to that CLI about what its agent may - # do, which is a value rather than a field that went missing: the rung it - # ran at was the CLI's own, and naming one here would be this bundle making - # up an answer the run never gave. - "permission": one.permission, # By name, and by name only: what an account runs a turn with is the one # thing a bundle must never carry. "provider": one.provider, - "goals": one.goals, - "person": one.person, "runs": one.spec, } for one in ran.agents ], + "envs": list(ran.envs), "called": _called(tree(epic)), "sessions": list(said), "backends": {name: _cli(name) for name in named}, diff --git a/src/hmz/runtime/flowing/__init__.py b/src/hmz/runtime/flowing/__init__.py index 0cfe0d3f..0ccd9808 100644 --- a/src/hmz/runtime/flowing/__init__.py +++ b/src/hmz/runtime/flowing/__init__.py @@ -1,40 +1,33 @@ -"""Everything humanize does to a flow: finding one, reading one, driving one, compiling one. +"""Everything humanize does to a flow: finding one, loading one, running one, and the drivers. A flow is content -- somebody else's repository, forked and edited -- and the whole of what it imports is :mod:`hmz.flows`: the protocols its agents and environments answer to, the decorator that makes it a flow, and the exceptions it can catch. This is the other side of -that line, and for now it holds two flow APIs. +that line. -The new one is written against :mod:`hmz.flows`. What a driver and the engine promise each -other is [spi.py](spi.py); what `-a`, `-e`, `-p` and `-b` say is [specs.py](specs.py); the -engine that defines, loads and runs flows is [engine.py](engine.py), with what a flow declares -read in [declaring.py](declaring.py), what it is handed in [viewing.py](viewing.py), what a -resumable run writes down in [journaling.py](journaling.py) and what a ref names in -[loading.py](loading.py); the drivers over coding agent CLIs and over machines are -[harnesses.py](harnesses.py) and [environments.py](environments.py); and in-memory stand-ins -for all of them, to test a flow with, are [fakes.py](fakes.py). +What a driver and the engine promise each other is [spi.py](spi.py); what `-a`, `-e`, `-p` and +`-b` say is [specs.py](specs.py); the engine that defines, loads and runs flows is +[engine.py](engine.py), with what a flow declares read in [declaring.py](declaring.py), what it +is handed in [viewing.py](viewing.py), what a resumable run writes down in +[journaling.py](journaling.py) and what a ref names in [loading.py](loading.py); the drivers +over coding agent CLIs and over machines are [harnesses.py](harnesses.py) and +[environments.py](environments.py); and in-memory stand-ins for all of them, to test a flow +with, are [fakes.py](fakes.py). Where flows come from and what each is called is +[verses.py](verses.py) and [finding.py](finding.py); the skills a flow named that live +somewhere else are fetched by [skills.py](skills.py). -The old one is written against :mod:`hmz._legacy_flows`, and goes when every way in has moved -over. Where flows come from and what each is called is [verses.py](verses.py) and -[finding.py](finding.py); what a flow says it drives, and what it takes for one flow to run -another, is [driving.py](driving.py); the two readings of a flow that refuse one before it can -cost anything are [checking.py](checking.py) and [proving.py](proving.py); an atlas is -compiled by [prophesying.py](prophesying.py) into the graph [prophecy.py](prophecy.py) -describes, and a run of one is walked by [stepping.py](stepping.py); the skills a flow named -that live somewhere else are fetched by [skills.py](skills.py). +The arrow points one way. Everything here may name the flow API, and it names nothing here at +the top of its files -- what a flow legitimately needs from this layer, which is defining a +flow, loading another and making an outworlder, is reached from there when the flow asks for +it. So a module that moves here moves without a flow anywhere noticing, which is the point of +the line being where it is. -The arrow points one way. Everything here may name either flow API, and neither names -anything here at the top of its file -- what a flow legitimately needs from this layer, which -is defining a flow, loading another and making an outworlder, is reached from there when the -flow asks for it. So a module that moves here moves without a flow anywhere noticing, which is -the point of the line being where it is. - -Nothing here drives a coding agent either. That is :mod:`hmz.coganchor`, which this is written -against and which names nothing here. +Nothing here drives a coding agent either. That is :mod:`hmz.coganchor`, which the harness +drivers are written against and which names nothing here. Everything is fetched when it is named, for the reason the layer above does it: a command line -that only lists the places flows come from must not pay for the `ast` of two readings, and -neither must a menu of flows. +that only lists the places flows come from must not pay for the engine, and neither must a +menu of flows. """ from __future__ import annotations @@ -42,36 +35,21 @@ from typing import TYPE_CHECKING if TYPE_CHECKING: - from .checking import Capability, Finding, briefed, catalogue, checked from .declaring import AgentRole, Declaration, EnvRole, Grant - from .driving import ( - Entry, - NotAFlow, - Place, - Running, - carries, - configures, - container, - declared, - drives, - load, - resumes, - running, - set_up, - wanted, - ) from .engine import ( Call, FlowImpl, LiveCall, Recorder, + current, define_flow, full_view, load_flow, new_outworlder, run_flow, + running, ) - from .environments import local_env, open_env + from .environments import local_env, open_env, probe from .fakes import ( FakeAgentDriver, FakeEnvDriver, @@ -82,48 +60,23 @@ from .finding import ( BUILTIN_AT, ENTRY, - PROPHECY, Offer, about, at, + builtin, entry, find, - foretold, fork, found, - held, inside, - loaded, offered, offers, - reading, + resolved, within, ) - from .harnesses import open_agent + from .harnesses import open_agent, open_outworlder from .journaling import FlowStateImpl, Journal - from .loading import FlowModule, Remote - from .prophecy import ( - Edge, - Node, - Prophecy, - Shape, - Shipped, - canonical, - digest, - kept, - told, - ) - from .prophecy import shipped as foreshipped - from .prophesying import Prophesied, is_atlas, prophesied - from .proving import ( - ALWAYS_DONE, - NEVER_DONE, - SILENT, - Outcome, - Proof, - Scenario, - proved, - ) + from .loading import FlowModule, Remote, forget from .skills import brought from .specs import ( AgentSpec, @@ -158,7 +111,6 @@ capabilities_of, default_result, ) - from .stepping import walking from .verses import ( FLOWS, LOCAL, @@ -174,7 +126,6 @@ __all__ = [ "AGENT_CAPABILITIES", - "ALWAYS_DONE", "BUILTIN_AT", "ENTRY", "ENV_CAPABILITIES", @@ -182,10 +133,7 @@ "HARNESS_CAPABILITIES", "LOCAL", "MINE", - "NEVER_DONE", "OFFICIAL", - "PROPHECY", - "SILENT", "USER", "AgentDriver", "AgentRole", @@ -195,10 +143,7 @@ "BoundHook", "BudgetSpecError", "Call", - "Capability", "Declaration", - "Edge", - "Entry", "EnvDriver", "EnvRole", "EnvSpec", @@ -208,7 +153,6 @@ "FakeEnvDriver", "FakeOutworlder", "FakeSession", - "Finding", "FlowImpl", "FlowModule", "FlowStateImpl", @@ -219,62 +163,37 @@ "Journal", "Limits", "LiveCall", - "Node", - "NotAFlow", "Offer", - "Outcome", "OutworlderDriver", "OutworlderView", "ParamSpecError", - "Place", "Placement", - "Proof", - "Prophecy", - "Prophesied", "Recorder", "Remote", - "Running", - "Scenario", "SessionHandle", "SessionView", - "Shape", - "Shipped", "Skill", "SpecError", "TurnRequest", "UsageSink", "about", "at", - "briefed", "brought", - "canonical", + "builtin", "capabilities_of", - "carries", - "catalogue", - "checked", - "configures", - "container", - "declared", + "current", "default_result", "define_flow", - "digest", - "drives", "entry", "find", "flowverses", - "foreshipped", - "foretold", + "forget", "fork", "found", "full_view", - "held", "holds", "inside", - "is_atlas", - "kept", - "load", "load_flow", - "loaded", "local_env", "nearest", "new_outworlder", @@ -282,32 +201,26 @@ "offers", "open_agent", "open_env", + "open_outworlder", "parse_agents", "parse_budget", "parse_duration", "parse_envs", "parse_params", - "prophesied", - "proved", - "reading", - "resumes", + "probe", + "resolved", "run_fake", "run_flow", "running", - "set_up", - "told", - "walking", - "wanted", "within", ] #: Which module each of them is written in. One entry per name this package offers, so that #: `from hmz.runtime.flowing import find` costs the module `find` is in rather than all of -#: them: the `ast` of two readings and every coding agent driver there is are behind some of -#: these, and a menu of flows must pay for none of it. +#: them: the engine and every coding agent driver there is are behind some of these, and a +#: menu of flows must pay for none of it. _WRITTEN = { "AGENT_CAPABILITIES": "hmz.runtime.flowing.spi", - "ALWAYS_DONE": "hmz.runtime.flowing.proving", "AgentDriver": "hmz.runtime.flowing.spi", "AgentRole": "hmz.runtime.flowing.declaring", "AgentSpec": "hmz.runtime.flowing.specs", @@ -317,12 +230,9 @@ "BoundHook": "hmz.runtime.flowing.spi", "BudgetSpecError": "hmz.runtime.flowing.specs", "Call": "hmz.runtime.flowing.engine", - "Capability": "hmz.runtime.flowing.checking", "Declaration": "hmz.runtime.flowing.declaring", "ENTRY": "hmz.runtime.flowing.finding", "ENV_CAPABILITIES": "hmz.runtime.flowing.spi", - "Edge": "hmz.runtime.flowing.prophecy", - "Entry": "hmz.runtime.flowing.driving", "EnvDriver": "hmz.runtime.flowing.spi", "EnvRole": "hmz.runtime.flowing.declaring", "EnvSpec": "hmz.runtime.flowing.specs", @@ -333,7 +243,6 @@ "FakeEnvDriver": "hmz.runtime.flowing.fakes", "FakeOutworlder": "hmz.runtime.flowing.fakes", "FakeSession": "hmz.runtime.flowing.fakes", - "Finding": "hmz.runtime.flowing.checking", "FlowImpl": "hmz.runtime.flowing.engine", "FlowModule": "hmz.runtime.flowing.loading", "FlowStateImpl": "hmz.runtime.flowing.journaling", @@ -347,30 +256,16 @@ "Limits": "hmz.runtime.flowing.spi", "LiveCall": "hmz.runtime.flowing.engine", "MINE": "hmz.runtime.flowing.verses", - "NEVER_DONE": "hmz.runtime.flowing.proving", - "Node": "hmz.runtime.flowing.prophecy", - "NotAFlow": "hmz.runtime.flowing.driving", "OFFICIAL": "hmz.runtime.flowing.verses", "Offer": "hmz.runtime.flowing.finding", - "Outcome": "hmz.runtime.flowing.proving", "OutworlderDriver": "hmz.runtime.flowing.spi", "OutworlderView": "hmz.runtime.flowing.viewing", - "PROPHECY": "hmz.runtime.flowing.finding", "ParamSpecError": "hmz.runtime.flowing.specs", - "Place": "hmz.runtime.flowing.driving", "Placement": "hmz.runtime.flowing.spi", - "Proof": "hmz.runtime.flowing.proving", - "Prophecy": "hmz.runtime.flowing.prophecy", - "Prophesied": "hmz.runtime.flowing.prophesying", "Recorder": "hmz.runtime.flowing.engine", "Remote": "hmz.runtime.flowing.loading", - "Running": "hmz.runtime.flowing.driving", - "SILENT": "hmz.runtime.flowing.proving", - "Scenario": "hmz.runtime.flowing.proving", "SessionHandle": "hmz.runtime.flowing.spi", "SessionView": "hmz.runtime.flowing.viewing", - "Shape": "hmz.runtime.flowing.prophecy", - "Shipped": "hmz.runtime.flowing.prophecy", "Skill": "hmz.runtime.flowing.spi", "SpecError": "hmz.runtime.flowing.specs", "TurnRequest": "hmz.runtime.flowing.spi", @@ -378,35 +273,22 @@ "UsageSink": "hmz.runtime.flowing.spi", "about": "hmz.runtime.flowing.finding", "at": "hmz.runtime.flowing.finding", - "briefed": "hmz.runtime.flowing.checking", "brought": "hmz.runtime.flowing.skills", - "canonical": "hmz.runtime.flowing.prophecy", + "builtin": "hmz.runtime.flowing.finding", "capabilities_of": "hmz.runtime.flowing.spi", - "carries": "hmz.runtime.flowing.driving", - "catalogue": "hmz.runtime.flowing.checking", - "checked": "hmz.runtime.flowing.checking", - "configures": "hmz.runtime.flowing.driving", - "container": "hmz.runtime.flowing.driving", - "declared": "hmz.runtime.flowing.driving", + "current": "hmz.runtime.flowing.engine", "default_result": "hmz.runtime.flowing.spi", "define_flow": "hmz.runtime.flowing.engine", - "digest": "hmz.runtime.flowing.prophecy", - "drives": "hmz.runtime.flowing.driving", "entry": "hmz.runtime.flowing.finding", "find": "hmz.runtime.flowing.finding", "flowverses": "hmz.runtime.flowing.verses", - "foretold": "hmz.runtime.flowing.finding", + "forget": "hmz.runtime.flowing.loading", "fork": "hmz.runtime.flowing.finding", "found": "hmz.runtime.flowing.finding", "full_view": "hmz.runtime.flowing.engine", - "held": "hmz.runtime.flowing.finding", "holds": "hmz.runtime.flowing.verses", "inside": "hmz.runtime.flowing.finding", - "is_atlas": "hmz.runtime.flowing.prophesying", - "kept": "hmz.runtime.flowing.prophecy", - "load": "hmz.runtime.flowing.driving", "load_flow": "hmz.runtime.flowing.engine", - "loaded": "hmz.runtime.flowing.finding", "local_env": "hmz.runtime.flowing.environments", "nearest": "hmz.runtime.flowing.verses", "new_outworlder": "hmz.runtime.flowing.engine", @@ -414,32 +296,20 @@ "offers": "hmz.runtime.flowing.finding", "open_agent": "hmz.runtime.flowing.harnesses", "open_env": "hmz.runtime.flowing.environments", + "open_outworlder": "hmz.runtime.flowing.harnesses", "parse_agents": "hmz.runtime.flowing.specs", "parse_budget": "hmz.runtime.flowing.specs", "parse_duration": "hmz.runtime.flowing.specs", "parse_envs": "hmz.runtime.flowing.specs", "parse_params": "hmz.runtime.flowing.specs", - "prophesied": "hmz.runtime.flowing.prophesying", - "proved": "hmz.runtime.flowing.proving", - "reading": "hmz.runtime.flowing.finding", - "resumes": "hmz.runtime.flowing.driving", + "probe": "hmz.runtime.flowing.environments", + "resolved": "hmz.runtime.flowing.finding", "run_fake": "hmz.runtime.flowing.fakes", "run_flow": "hmz.runtime.flowing.engine", - "running": "hmz.runtime.flowing.driving", - "set_up": "hmz.runtime.flowing.driving", - "told": "hmz.runtime.flowing.prophecy", - "walking": "hmz.runtime.flowing.stepping", - "wanted": "hmz.runtime.flowing.driving", + "running": "hmz.runtime.flowing.engine", "within": "hmz.runtime.flowing.finding", } -#: The one name this package offers under something other than its own. `shipped` is what a -#: flow's directory ships a compiled prophecy as; `shipped` is also what a session carries of -#: the skills a flow brought. Two of one word in one namespace is a reader having to be told -#: which is meant, so the prophecy's is offered here as `foreshipped` and is `shipped` in the -#: module it is written in, where there is only ever one of it. -_AS = {"foreshipped": "shipped"} - def __getattr__(name: str) -> object: """Hands through what this package offers, out of the module it is written in. @@ -453,11 +323,11 @@ def __getattr__(name: str) -> object: Raises: AttributeError: If nothing here is called that, as for any other module. It is also what sends Python looking for a module of that name beside this one, which is how - `from hmz.runtime.flowing import checking` goes on being the reading rather than this. + `from hmz.runtime.flowing import engine` goes on being the module rather than this. """ from importlib import import_module where_ = _WRITTEN.get(name) if where_ is None: raise AttributeError(f"module {__name__!r} has no attribute {name!r}") - return getattr(import_module(where_), _AS.get(name, name)) + return getattr(import_module(where_), name) diff --git a/src/hmz/runtime/flowing/checking.py b/src/hmz/runtime/flowing/checking.py deleted file mode 100644 index d42bdcab..00000000 --- a/src/hmz/runtime/flowing/checking.py +++ /dev/null @@ -1,2795 +0,0 @@ -"""The static read of a flow's legality: what will not run, said before anything runs it. - -:mod:`hmz.runtime.flowing.driving` refuses a flow as it loads it -- the wrong arity, a -moment no agent can run -- but loading a flow means running its file, and the flow most worth -checking is one nobody has read yet: generated, fetched, forked and edited. This is the -reading that executes -nothing. Pure `ast` over every Python file the flow's directory holds, answering with findings -rather than raising, so that whatever asked can say everything that is wrong at once. - -What a severity means is one line apiece. An error is a flow that cannot run, cannot be -answered, or cannot end -- something no run of it survives. A warning is a flow that runs, -and a run of it that may be regretted: a loop with no bound of its own, a shaped answer read -without a guard, a config that takes anything. - -And what the reading does not claim is said as plainly. Every rule is the proof of an -absence -- no exit in this loop, no bound in this function, no guard on this name -- worked -out one function at a time. Nothing here proves an exit reachable or a bound tight, and -nothing follows a value through a call: a flow that keeps its loop in one function and its -bound in another is a flow this reading trusts, and a checker that guessed further would -refuse flows that run. -""" - -from __future__ import annotations - -import ast -import contextlib -from dataclasses import dataclass, field -from pathlib import Path -from typing import TYPE_CHECKING, Literal, NamedTuple, Protocol - -from hmz._legacy_flows import Agent, Driven, Person, Session -from hmz.coganchor import places - -if TYPE_CHECKING: - import os - from collections.abc import Iterable, Iterator, Mapping, Sequence - -__all__ = [ - "Capability", - "Finding", - "briefed", - "catalogue", - "checked", - "offered", - "surface", -] - - -class Finding(NamedTuple): - """One thing the reading of a flow found, where it found it. - - Attributes: - code: Which rule found it, as one hyphenated word -- `dead-loop`, `unknown-ask`. - severity: What finding it is. An error is a flow that cannot run, cannot be answered - or cannot end; a warning is a flow that runs and may be regretted. - where: The file it is in. - line: The line, 1-based, or 0 for a finding about the whole file. - said: What is wrong, said the way `NotAFlow` says it. - """ - - code: str - severity: Literal["error", "warning"] - where: Path - line: int - said: str - - -def surface(protocol: type) -> frozenset[str]: - """What one of the flow-facing interfaces asks for, by name. - - The one reading of a protocol's members, shared by the checker's own rules and by the - tests that hold the drivers to the same interfaces -- two readings of what an agent - answers to would be two readings to drift apart. - - Args: - protocol: The interface. - - Returns: - One name per member it declares, its own and whatever it is itself an interface of. - `__call__` is among them where it is declared, an agent and a session both being things - a flow calls; the rest of the dunders are Python's and are not part of the contract. - """ - said: set[str] = set() - for one in protocol.__mro__: - if one in (object, Protocol): - continue - held = set(vars(one)) | set(getattr(one, "__annotations__", {})) - said.update( - name for name in held if not name.startswith("_") or name == "__call__" - ) - return frozenset(said) - - -def offered() -> frozenset[str]: - """Every name a flow may import from `hmz._legacy_flows`, which is the whole of its vocabulary. - - What the module says it offers rather than what happens to be reachable on it: a name - that works today because a submodule leaked it is a name the next release takes away. - - Returns: - The names: what `__all__` declares, what is handed through by name from the layers a - flow does not import, and the two modules handed through whole. - """ - # The package's own tables, read by the package's own checker: private to every - # flow, and one copy rather than a second one kept here to drift. - from hmz._legacy_flows import ( - _ELSEWHERE, # pyright: ignore[reportPrivateUsage] - _MODULES, # pyright: ignore[reportPrivateUsage] - ) - from hmz._legacy_flows import __all__ as declared - - return frozenset(declared) | frozenset(_ELSEWHERE) | frozenset(_MODULES) - - -def checked(flow: str | os.PathLike[str]) -> tuple[Finding, ...]: - """Reads a flow without running it, and answers with everything that reading found. - - Every Python file the flow's directory holds is read -- the entry point and whatever it - imports beside it -- except what is under its `skills/`, which is content for the agents - rather than code this process runs. Nothing is imported and nothing is executed, so this - is safe to point at a flow nobody has read: what running the file would refuse is the - second reading, :mod:`hmz.runtime.flowing.proving`, which runs it in a process of its own. - - A flow marked `@atlas` gets the stricter reading rather than this one, which is - :func:`hmz.runtime.flowing.prophesying.prophesied`: an atlas is a flow whose body is - compiled, so the rules that read a body as a program would be reading it as something it - is not. - - Args: - flow: The flow: its directory, or the Python file a single-file flow is. - - Returns: - One finding per thing found, in file order, and nothing at all for a flow this reading - has nothing to say about -- which is not a proof, only a reading with nothing to say. - """ - whole = _whole(flow) - if whole.compiled: - from .prophesying import prophesied - - return prophesied(flow, whole=whole).findings - return _rules(whole) - - -class _Whole(NamedTuple): - """One flow's files, parsed: what both readings start from. - - Attributes: - entry: Where the flow's entry point is. - read: One per file that parsed, in file order. - entered: The entry point's own file, or None where it could not be read. - found: What reading the files found before any rule ran -- a file that will not - parse, a directory that holds no flow. - compiled: Whether the entry point holds an atlas, which is a flow to compile rather - than a flow to read as a program. - declared: The bodies an atlas compiles, by the identity of the `ast` node each is. - By identity rather than by name: a class beside an atlas with a method of the same - name is ordinary Python, and one skipped for sharing a spelling is one nothing - reads at all. - """ - - entry: Path - read: list[_Read] - entered: _Read | None - found: list[Finding] - compiled: bool - declared: frozenset[int] - - -def _whole(flow: str | os.PathLike[str]) -> _Whole: - """Parses every Python file one flow holds, and says which kind of flow it is. - - Args: - flow: The flow: its directory, or the Python file a single-file flow is. - - Returns: - The files and what parsing them found. Nothing is imported and nothing is executed. - """ - from .finding import ENTRY - - at = Path(flow) - if at.is_dir(): - entry = at / ENTRY - files = [ - one - for one in sorted(at.rglob("*.py")) - if "__pycache__" not in one.relative_to(at).parts - # The skills are content: one directory per skill, laid out the way every one - # of these CLIs reads a skill in, and nothing in one is imported by the flow. - and one.relative_to(at).parts[0] != "skills" - ] - else: - entry = at - files = [at] if at.is_file() else [] - if not entry.is_file(): - return _Whole( - entry, - [], - None, - [ - Finding( - "not-a-flow", - "error", - at, - 0, - "no flow to read: a flow is a directory with an __init__.py in it", - ) - ], - compiled=False, - declared=frozenset(), - ) - - found: list[Finding] = [] - read: list[_Read] = [] - for one in files: - held = _parsed(one) - if isinstance(held, Finding): - found.append(held) - else: - read.append(held) - entered = next((one for one in read if one.where == entry), None) - compiled = entered is not None and any(one.atlas for one in entered.marks) - return _Whole( - entry, - read, - entered, - found, - compiled=compiled, - declared=frozenset( - id(one.node) for said in read for one in said.marks if one.atlas - ), - ) - - -def _rules(whole: _Whole) -> tuple[Finding, ...]: - """Every rule that reads a flow as a program, run over the files that parsed. - - What an atlas compiles is left out: those bodies are declarations rather than programs, - and would be refused as both -- an `if` with no `elif` is a branch there, and a `while` - with no `break` is an edge back to a node. - - Args: - whole: What :func:`_whole` read. - - Returns: - One finding per thing found, in file order. - """ - found = list(whole.found) - entered = whole.entered - if entered is not None and not any(one.marks for one in whole.read): - found.append( - Finding( - "not-a-flow", - "error", - whole.entry, - 0, - "nothing in it is marked @flow() -- a flow is a function marked with it, " - "which is how a file says which of the functions in it is one", - ) - ) - if entered is not None and ast.get_docstring(entered.tree) is None: - found.append( - Finding( - "unsaid-flow", - "warning", - whole.entry, - 0, - "the flow says nothing about itself -- the first line of this file's " - "docstring is what every list of flows shows for it", - ) - ) - - # What the flow declared about moments anywhere in its files, for the hooks it hangs: - # a place annotated in the entry point covers a hook hung in the module beside it. - declared = frozenset( - moment for one in whole.read for moment in one.moments_declared - ) - asks = _Asks() - for one in whole.read: - found.extend(_imports(one)) - found.extend(_marks(one)) - found.extend(_declares(one)) - found.extend(_hooks(one, declared)) - found.extend(_functions(one, asks, whole.declared)) - return tuple(found) - - -# --------------------------------------------------------------------------------------- -# Reading one file: what it imports, what it marks, and what it declares. -# --------------------------------------------------------------------------------------- - -#: The kinds of thing a flow drives, each the name of one flow-facing interface. What a -#: tracked name may be asked is read off the interface itself, so the checker and the -#: drivers are held to one surface. -_KINDS: dict[str, type] = { - "agent": Agent, - "person": Person, - "session": Session, - "driven": Driven, -} - -#: How the local names for those interfaces read where a flow imports them. -_PROTOCOLS = { - "Agent": "agent", - "Person": "person", - "Session": "session", - "Driven": "driven", -} - -#: The two marks that make a function a node of a prophecy, by the name `hmz._legacy_flows` offers -#: each under. -_NODES = ("mind", "logic") - -#: What an element of a plain tuple of agents may still be asked by name: the tuple's own -#: two methods. A named place is a field of the NamedTuple the flow declared instead. -_OF_A_TUPLE = frozenset({"count", "index"}) - - -class _Mark(NamedTuple): - """One function a file marked as a flow, and what the mark said. - - Attributes: - node: The function. - name: What the mark called it inside its file, and "" for the one the file holds - under its own name. - resumable: Whether it says it can be picked up where the last run of it left off, - which an atlas always says. - atlas: Whether it was marked `@atlas` rather than `@flow` -- a flow whose body is - read rather than run, and which `prophesying.py` compiles. - """ - - node: ast.FunctionDef | ast.AsyncFunctionDef - name: str - resumable: bool - atlas: bool = False - - -class _Node(NamedTuple): - """One function a file marked `@mind` or `@logic`, and what the mark said. - - Attributes: - node: The function, which is what its parameters and its answer are read off. - kind: Which of the two it is. - rerun: Whether a run picked up again runs it again where the last one stopped inside - it, or steps past it. - """ - - node: ast.FunctionDef | ast.AsyncFunctionDef - kind: str - rerun: bool - - -@dataclass -class _Read: - """One file, parsed, and what one pass over its top collected.""" - - where: Path - tree: ast.Module - #: The local names of :func:`hmz._legacy_flows.flow`, `Moment` and pydantic's `Field`. - flow_alias: set[str] = field(default_factory=set[str]) - moment_alias: set[str] = field(default_factory=set[str]) - field_alias: set[str] = field(default_factory=set[str]) - #: And of :func:`hmz._legacy_flows.atlas`, :func:`hmz._legacy_flows.sub` and - #: :func:`hmz._legacy_flows.load`: - #: the mark that says a flow is compiled, the one way an atlas reaches another, and the - #: one way an ordinary flow does -- which is the call an atlas may not write. - atlas_alias: set[str] = field(default_factory=set[str]) - sub_alias: set[str] = field(default_factory=set[str]) - load_alias: set[str] = field(default_factory=set[str]) - #: Local name -> which kind of node, for `mind` and `logic` as this file imports them. - node_alias: dict[str, str] = field(default_factory=dict[str, str]) - #: Local name -> which interface, for every flow-facing interface the file imports. - proto: dict[str, str] = field(default_factory=dict[str, str]) - #: The NamedTuple and pydantic model classes the file itself declares. - crews: dict[str, ast.ClassDef] = field(default_factory=dict[str, ast.ClassDef]) - models: dict[str, ast.ClassDef] = field(default_factory=dict[str, ast.ClassDef]) - #: Names bound only under `if TYPE_CHECKING:`, which a running flow cannot read. - unread: set[str] = field(default_factory=set[str]) - #: Every name the file binds at module level as it runs, which excuses the above. - bound: set[str] = field(default_factory=set[str]) - marks: list[_Mark] = field(default_factory=list["_Mark"]) - #: Every `Moment.X` written inside an annotation, which is a flow declaring a need. - moments_declared: set[str] = field(default_factory=set[str]) - #: The functions this file marked `@mind` or `@logic`, by the name it declares each - #: under, and what each mark said. - nodes: dict[str, _Node] = field(default_factory=dict[str, "_Node"]) - #: And the atlases of other files it named, `: ` apiece, which is a - #: module-level `review = sub("official/review")`. - subs: dict[str, str] = field(default_factory=dict[str, str]) - - -def _parsed(where: Path) -> _Read | Finding: - """One file read into a tree, or the finding that it could not be. - - Args: - where: The file. - - Returns: - What was read, or an `unread` error: a file that will not parse is a flow that will - not load, said here rather than left for the loading to hit. - """ - try: - tree = ast.parse(where.read_text(encoding="utf-8")) - except (OSError, UnicodeDecodeError, SyntaxError, ValueError) as why: - line = getattr(why, "lineno", 0) or 0 - return Finding( - "unread", - "error", - where, - line, - f"nothing here can be read as Python -- {why}", - ) - read = _Read(where, tree) - _collected(read, tree.body, type_checking=False) - # Bound under TYPE_CHECKING and nowhere else: a name a type checker reads and a - # running flow cannot, which is the one thing `unread-annotation` is about. - read.unread -= read.bound - for node in ast.walk(tree): - if isinstance(node, (ast.AnnAssign, ast.arg)) and node.annotation is not None: - read.moments_declared.update(_moments_in(node.annotation, read)) - return read - - -def _collected(read: _Read, body: list[ast.stmt], *, type_checking: bool) -> None: - """Walks one file's statements for what the rules read off its top. - - Args: - read: What is being collected into. - body: The statements, at whatever depth the walk has reached. - type_checking: Whether these statements are under `if TYPE_CHECKING:`, where a name - is bound for a type checker and for nothing that runs. - """ - for node in body: - if isinstance(node, ast.ImportFrom) and node.module == "hmz._legacy_flows": - for alias in node.names: - bound = alias.asname or alias.name - if type_checking: - read.unread.add(bound) - continue - read.bound.add(bound) - if alias.name == "flow": - read.flow_alias.add(bound) - elif alias.name == "atlas": - read.atlas_alias.add(bound) - elif alias.name == "sub": - read.sub_alias.add(bound) - elif alias.name == "load": - read.load_alias.add(bound) - elif alias.name in _NODES: - read.node_alias[bound] = alias.name - elif alias.name == "Moment": - read.moment_alias.add(bound) - elif alias.name in _PROTOCOLS: - read.proto[bound] = _PROTOCOLS[alias.name] - elif isinstance(node, (ast.Import, ast.ImportFrom)): - for alias in node.names: - bound = (alias.asname or alias.name).split(".")[0] - (read.unread if type_checking else read.bound).add(bound) - if ( - isinstance(node, ast.ImportFrom) - and node.module == "pydantic" - and alias.name == "Field" - and not type_checking - ): - read.field_alias.add(alias.asname or alias.name) - elif isinstance(node, (ast.Assign, ast.AnnAssign)) and not type_checking: - targets = node.targets if isinstance(node, ast.Assign) else [node.target] - read.bound.update(one.id for one in targets if isinstance(one, ast.Name)) - _named_sub(read, targets, node.value) - elif isinstance(node, ast.ClassDef) and not type_checking: - read.bound.add(node.name) - # By what each base is called at the tip: `pydantic.BaseModel` and - # `typing.NamedTuple` are the same two classes reached the other way, and - # one read at the root would be the module's name and neither of them. - bases = {_tip(base) for base in node.bases} - if "NamedTuple" in bases: - read.crews[node.name] = node - elif "BaseModel" in bases or bases & set(read.models): - read.models[node.name] = node - elif isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)): - if not type_checking: - read.bound.add(node.name) - mark = _marked(node, read) - if mark is not None and not type_checking: - read.marks.append(mark) - held = _noded(node, read) - if held is not None and not type_checking: - read.nodes[node.name] = held - elif isinstance(node, ast.If): - under = type_checking or _root(node.test) == "TYPE_CHECKING" - _collected(read, node.body, type_checking=under) - _collected(read, node.orelse, type_checking=type_checking) - elif isinstance(node, (ast.Try, ast.With)): - _collected(read, node.body, type_checking=type_checking) - - -def _named_sub( - read: _Read, targets: Sequence[ast.expr], value: ast.expr | None -) -> None: - """Records `review = sub("official/review")`, which is one supernode named. - - Args: - read: What is being collected into. - targets: What the statement assigns to. - value: What it assigns, which is only ever read where it is that call. - """ - if not ( - isinstance(value, ast.Call) - and isinstance(value.func, ast.Name) - and value.func.id in read.sub_alias - and len(value.args) == 1 - and isinstance(value.args[0], ast.Constant) - and isinstance(value.args[0].value, str) - ): - return - for one in targets: - if isinstance(one, ast.Name): - read.subs[one.id] = value.args[0].value - - -def _marked(node: ast.FunctionDef | ast.AsyncFunctionDef, read: _Read) -> _Mark | None: - """The mark on one function, where it carries one. - - Both marks: `@flow` and `@atlas` both make a function a flow, and everything that reads - a flow off a file reads both -- an atlas that was invisible here would be a flow nothing - could list, name or refuse. Which of the two it was is on the mark, for the one reading - that has to know. - - Args: - node: The function. - read: The file it is in, for what the decorator is called there. - - Returns: - The mark, or None for a function that is not a flow. - """ - for one in node.decorator_list: - called = one.func if isinstance(one, ast.Call) else one - if not isinstance(called, ast.Name): - continue - atlas = called.id in read.atlas_alias - if not atlas and called.id not in read.flow_alias: - continue - if not isinstance(one, ast.Call): - return _Mark(node, "", resumable=atlas, atlas=atlas) - name = "" - # An atlas can always be picked up: a prophecy is a list of nodes with an answer - # apiece, so what a run of one has done is something the run writes down itself. - resumable = atlas - for said in one.keywords: - if said.arg == "name" and isinstance(said.value, ast.Constant): - name = str(said.value.value) - elif ( - said.arg == "resumable" - and not atlas - and isinstance(said.value, ast.Constant) - ): - resumable = bool(said.value.value) - return _Mark(node=node, name=name, resumable=resumable, atlas=atlas) - return None - - -def _noded(node: ast.FunctionDef | ast.AsyncFunctionDef, read: _Read) -> _Node | None: - """The `@mind` or `@logic` mark on one function, where it carries one. - - Args: - node: The function. - read: The file it is in, for what the decorator is called there. - - Returns: - What the mark said, or None for a function that is not a node. - """ - for one in node.decorator_list: - called = one.func if isinstance(one, ast.Call) else one - if not isinstance(called, ast.Name) or called.id not in read.node_alias: - continue - rerun = True - if isinstance(one, ast.Call): - for said in one.keywords: - if said.arg == "rerun" and isinstance(said.value, ast.Constant): - rerun = bool(said.value.value) - return _Node(node=node, kind=read.node_alias[called.id], rerun=rerun) - return None - - -def _root(node: ast.expr) -> str: - """The name at the root of one expression, or "" where there is none.""" - while isinstance(node, ast.Attribute): - node = node.value - return node.id if isinstance(node, ast.Name) else "" - - -def _tip(node: ast.expr) -> str: - """The name at the tip of one dotted expression -- `Literal` of `typing.Literal`.""" - if isinstance(node, ast.Attribute): - return node.attr - return node.id if isinstance(node, ast.Name) else "" - - -def _moments_in(annotation: ast.expr, read: _Read) -> set[str]: - """Every moment one annotation names, which is a place saying what it needs.""" - return { - node.attr - for node in ast.walk(annotation) - if isinstance(node, ast.Attribute) - and isinstance(node.value, ast.Name) - and node.value.id in read.moment_alias - } - - -# --------------------------------------------------------------------------------------- -# The import rules: one import, and only the names it offers. -# --------------------------------------------------------------------------------------- - - -def _imports(read: _Read) -> Iterator[Finding]: - """What one file imports of humanize's, held to the one import a flow writes. - - Args: - read: The file. - - Yields: - A `foreign-import` error per import of a module of humanize's own that is not - `hmz._legacy_flows`, and an `unknown-name` error per name asked of `hmz._legacy_flows` - that it does - not offer. - """ - offers: frozenset[str] | None = None - for node in ast.walk(read.tree): - if isinstance(node, ast.Import): - for alias in node.names: - if ( - alias.name.split(".")[0] == "hmz" - and alias.name != "hmz._legacy_flows" - ): - yield Finding( - "foreign-import", - "error", - read.where, - node.lineno, - f"a flow imports hmz._legacy_flows and nothing else of humanize's -- " - f"{alias.name} is humanize's own business, and a flow that names " - "it breaks whenever humanize moves it", - ) - elif isinstance(node, ast.ImportFrom) and node.module: - said = node.module - if said.split(".")[0] != "hmz": - continue - if said != "hmz._legacy_flows": - yield Finding( - "foreign-import", - "error", - read.where, - node.lineno, - f"a flow imports hmz._legacy_flows and nothing else of humanize's -- " - f"{said} is humanize's own business, and a flow that names it " - "breaks whenever humanize moves it", - ) - continue - if offers is None: - offers = offered() - for alias in node.names: - if alias.name not in offers: - yield Finding( - "unknown-name", - "error", - read.where, - node.lineno, - f"hmz._legacy_flows does not offer {alias.name!r} -- what a flow may " - "import from it is what it hands through, and a name it does " - "not hold fails at the first run", - ) - - -# --------------------------------------------------------------------------------------- -# The mark rules: what the entry point declares, read as the loader will read it. -# --------------------------------------------------------------------------------------- - -#: How many arguments a resumable flow's entry point takes at the least: the agents, the -#: task, and somewhere for what the last run of it wrote to be handed back in. -_WITH_A_STATE = 3 - -#: The one sentence the loader says about a flow that does not state its arity, said here -#: too so that the two readings refuse the same flow in the same words. -_UNSIZED = ( - "a flow is a function marked @flow() taking (agents, task), whose agents are " - "annotated with a tuple of a fixed length -- how many agents the flow drives -- or " - "with a NamedTuple of them, which also says what each is for" -) - - -def _marks(read: _Read) -> Iterator[Finding]: - """What each flow one file marks says about itself, held to what a run needs said. - - Args: - read: The file. - - Yields: - `unsized-agents` and `unread-annotation` errors for an arity a run cannot read back, - `stateless-resume` for a flow that says it can be picked up and takes nothing to be - picked up with, `twice-named` for two flows under one name, `state-kept` for kept - state nothing ever clears, and the config findings for the model an entry declares. - """ - named: set[str] = set() - configured: set[str] = set() - for mark in read.marks: - if mark.name in named: - yield Finding( - "twice-named", - "warning", - read.where, - mark.node.lineno, - f"two flows here are both {_said_name(mark.name)} -- the first of them " - "wins, and the second can never be run", - ) - named.add(mark.name) - yield from _sized(mark, read) - # An atlas is resumable without taking a dict to be resumed with: what a run of one - # has done is which of its nodes have answered, which the run writes down itself. - if not mark.atlas: - yield from _resumes(mark, read) - for model in _settings(mark.node, read): - if model not in configured: - configured.add(model) - yield from _configured(read.models[model], read) - - -def _said_name(name: str) -> str: - """How a flow's name reads in a finding about two of them.""" - return f"called {name!r}" if name else "the one their file holds under its own name" - - -def _declares(read: _Read) -> Iterator[Finding]: - """What a flow says its agents run at, held to what can actually be said. - - Read wherever it is written rather than only on the entry point: a flow declares its - places in a NamedTuple beside the function as often as inside its annotation, and both - are the same declaration reaching the same agent. - - Only a rung written out as a literal is held to anything. A place that writes - `AgentDefaults(permission=UNSAID)` writes a name rather than a constant, and what a name - stands for is something this file cannot know without running the flow it is checking -- - which is the one thing a check must not do. So that spelling goes by unread, deliberately - rather than by oversight: what escapes here is refused all the same, by the config itself, - on the first line of the run. - - Args: - read: The file. - - Yields: - `unknown-permission` for a rung no backend has a word for, which is a run refused as - the flow loads, and `goals-both-ways` for a place declared under a goal and without - goals at once -- the flow saying two things about one agent, of which only one can be - done. - """ - # Both read off the package rather than written down again here, and composed into the - # one tuple the config validates against: the rungs are the ladder a flow may declare, - # and the silence is the answer above all of them, which is not a rung and is not in - # `PERMISSIONS`. Composed at this call site rather than imported as the config's own - # `_SAYABLE`, because that name is private to the module that settles these and a checker - # reaching across for it would be one more thing to keep in step. - from hmz.coganchor.agents import PERMISSIONS, UNSAID - - sayable = (*PERMISSIONS, UNSAID) - for node in ast.walk(read.tree): - if isinstance(node, ast.Call) and _tip(node.func) == "AgentDefaults": - for said in node.keywords: - if ( - said.arg == "permission" - and isinstance(said.value, ast.Constant) - and said.value.value not in sayable - ): - yield Finding( - "unknown-permission", - "error", - read.where, - node.lineno, - f"{said.value.value!r} is no rung there is -- what an agent may " - f"do is one of {', '.join(PERMISSIONS)}, or {UNSAID!r} for a place " - "that would rather humanize said nothing and left the CLI wherever " - "its own headless run leaves it, and a flow declaring anything else " - "is refused before its first turn", - ) - if ( - isinstance(node, ast.Subscript) - and _tip(node.value) == "Annotated" - and _goal_written(node) - and _goals_off(node) - ): - yield Finding( - "goals-both-ways", - "error", - read.where, - node.lineno, - "this place is run under a goal and declared without goals -- a required " - "goal is a goal, so drop one of the two rather than leaving the flow to " - "say which it meant", - ) - - -def _goal_written(node: ast.Subscript) -> bool: - """Whether one `Annotated` place says its agent is run under a backend goal.""" - return any(_tip(one) == "Goal" for one in _elements(node.slice)[1:]) - - -def _goals_off(node: ast.Subscript) -> bool: - """Whether one `Annotated` place declares that its agent has no goals.""" - return any( - isinstance(one, ast.Call) - and _tip(one.func) == "AgentDefaults" - and any( - said.arg == "goals" - and isinstance(said.value, ast.Constant) - and said.value.value is False - for said in one.keywords - ) - for one in _elements(node.slice)[1:] - ) - - -def _sized(mark: _Mark, read: _Read) -> Iterator[Finding]: - """Whether one flow states how many agents it drives where a run can read it back.""" - args = mark.node.args - params = [*args.posonlyargs, *args.args] - kind = params[0].annotation if params else None - if kind is None: - yield Finding("unsized-agents", "error", read.where, mark.node.lineno, _UNSIZED) - return - unread = sorted(_names_in(kind) & read.unread) - if unread: - yield Finding( - "unread-annotation", - "error", - read.where, - mark.node.lineno, - f"the flow's agents cannot be read here ({', '.join(unread)} is imported " - "under TYPE_CHECKING) -- import what the annotation names at runtime, so " - "the count it states can be checked", - ) - return - said = _unquoted(kind) - if said is None: - return - if isinstance(said, ast.Name) and said.id == "tuple": - yield Finding( - "unsized-agents", - "error", - read.where, - mark.node.lineno, - _UNSIZED, - ) - return - if ( - isinstance(said, ast.Subscript) - and _root(said.value) == "tuple" - and any( - isinstance(one, ast.Constant) and one.value is Ellipsis - for one in _elements(said.slice) - ) - ): - yield Finding( - "unsized-agents", - "error", - read.where, - mark.node.lineno, - _UNSIZED, - ) - - -def _resumes(mark: _Mark, read: _Read) -> Iterator[Finding]: - """Whether a flow that says it can be picked up takes anything to be picked up with.""" - if not mark.resumable: - return - args = mark.node.args - params = [*args.posonlyargs, *args.args] - taken = len(params) - third = params[2].annotation if taken >= _WITH_A_STATE else None - settles = third is not None and bool(_names_in(third) & set(read.models)) - if taken < _WITH_A_STATE or (taken == _WITH_A_STATE and settles): - yield Finding( - "stateless-resume", - "error", - read.where, - mark.node.lineno, - "the flow says it can be picked up, and takes nothing to be picked up with " - "-- a resumable flow is handed a dict as its last argument, holding what it " - "wrote there last time", - ) - return - yield from _kept(mark, read, params[-1].arg) - - -def _kept(mark: _Mark, read: _Read, state: str) -> Iterator[Finding]: - """Whether kept state something writes is state anything ever clears. - - Args: - mark: The resumable flow. - read: Its file. - state: What its entry point calls the dict it is handed. - """ - held = {state} - wrote = 0 - cleared = False - for node in ast.walk(mark.node): - if isinstance(node, ast.Assign) and len(node.targets) == 1: - target = node.targets[0] - if isinstance(target, ast.Name) and any( - isinstance(one, ast.Name) and one.id in held - for one in ast.walk(node.value) - ): - held.add(target.id) - elif isinstance(node, ast.Subscript) and isinstance(node.ctx, ast.Store): - if _root(node.value) in held: - wrote = wrote or node.lineno - elif isinstance(node, ast.Delete): - # `del state[what]` is the same thing said the other way round: a flow that - # emptied what it kept is a flow the next run here opens on nothing. - cleared = cleared or any( - isinstance(one, ast.Subscript) and _root(one.value) in held - for one in node.targets - ) - elif ( - isinstance(node, ast.Call) - and isinstance(node.func, ast.Attribute) - and _root(node.func.value) in held - ): - if node.func.attr in {"update", "setdefault"}: - wrote = wrote or node.lineno - elif node.func.attr in {"clear", "pop", "popitem"}: - cleared = True - if wrote and not cleared and _finishes(mark.node): - yield Finding( - "state-kept", - "warning", - read.where, - wrote, - "the flow writes its kept state and never clears it -- a run that is over " - "leaves what the next run here opens on, so a loop that has ended clears " - "what it kept", - ) - - -def _finishes(node: ast.FunctionDef | ast.AsyncFunctionDef) -> bool: - """Whether this flow has any way of ending that is the flow deciding it is done. - - A loop with no `break`, no `return` and no `raise` in it cannot decide anything: what ends - it is the run's allowance being spent, and a run stopped for that is a run to pick up - rather than one that is over. So there is nothing for it to clear, and telling it to clear - would be telling it to throw away the state that makes picking it up worth doing. - - Args: - node: The entry point. - - Returns: - Whether some loop in it can be left from inside. A flow with no constant-true loop at - all falls off its own end, which is a flow that finished. - """ - loops = [ - loop - for loop in _whiles(node.body) - if isinstance(loop.test, ast.Constant) and loop.test.value - ] - return not loops or any(_exits(loop.body, ()) for loop in loops) - - -def _settings( - node: ast.FunctionDef | ast.AsyncFunctionDef, read: _Read -) -> Iterator[str]: - """The config models one entry point declares, of the ones its own file holds.""" - args = node.args - for param in [*args.posonlyargs, *args.args][2:]: - if param.annotation is not None: - yield from ( - name for name in _names_in(param.annotation) if name in read.models - ) - - -def _configured(model: ast.ClassDef, read: _Read) -> Iterator[Finding]: - """One config model, held to refusing what it does not take and saying what it does. - - Args: - model: The model a flow's entry point says it can be set up with. - read: The file it is declared in. - - Yields: - A `loose-config` warning for one that takes anything, and an `unsaid-field` warning - per field that says nothing about itself -- the descriptions are what whoever sets - the flow up is shown, and a field without one is a question nobody can answer. - """ - if not _strict(model, read, set()): - yield Finding( - "loose-config", - "warning", - read.where, - model.lineno, - "the config takes anything -- set model_config to extra: forbid or frozen: " - "True, so a setting that is misspelled is refused rather than quietly " - "ignored", - ) - for node in model.body: - if not isinstance(node, ast.AnnAssign) or not isinstance(node.target, ast.Name): - continue - name = node.target.id - if name.startswith("_") or _root(node.annotation) == "ClassVar": - continue - bare = node.value is None - unsaid = ( - isinstance(node.value, ast.Call) - and _root(node.value.func) in read.field_alias - and not any(one.arg == "description" for one in node.value.keywords) - ) - if bare or unsaid: - yield Finding( - "unsaid-field", - "warning", - read.where, - node.lineno, - f"the config field {name!r} says nothing about itself -- give it a " - "Field(description=...), which is what whoever sets the flow up is " - "shown", - ) - - -def _strict(model: ast.ClassDef, read: _Read, seen: set[str]) -> bool: - """Whether one config model refuses what it does not take, its local bases included.""" - seen.add(model.name) - for node in model.body: - if ( - isinstance(node, ast.Assign) - and any( - isinstance(one, ast.Name) and one.id == "model_config" - for one in node.targets - ) - and isinstance(node.value, ast.Dict) - ): - said = { - key.value: value.value - for key, value in zip(node.value.keys, node.value.values, strict=True) - if isinstance(key, ast.Constant) and isinstance(value, ast.Constant) - } - if said.get("extra") == "forbid" or said.get("frozen") is True: - return True - return any( - _strict(read.models[base], read, seen) - for base in {_root(one) for one in model.bases} - if base in read.models and base not in seen - ) - - -# --------------------------------------------------------------------------------------- -# The hook rule: a moment only some backends run is a moment the flow says it needs. -# --------------------------------------------------------------------------------------- - - -def _hooks(read: _Read, declared: frozenset[str]) -> Iterator[Finding]: - """Every moment one file hangs a hook on, held to the moments the flow declared. - - Args: - read: The file. - declared: Every moment named in an annotation anywhere in the flow, which is how a - place says what the agent filling it has to run. - - Yields: - An `unsaid-moment` warning per hook hung on a moment only some backends reach that - no place declares: the run finds out from `Unhooked`, mid-flow, where a declaration - would have refused the agent before its first turn. - """ - everywhere: frozenset[str] | None = None - for node in ast.walk(read.tree): - if not ( - isinstance(node, ast.Call) - and isinstance(node.func, ast.Attribute) - and node.func.attr == "on" - and node.args - ): - continue - moment = node.args[0] - if not ( - isinstance(moment, ast.Attribute) - and isinstance(moment.value, ast.Name) - and moment.value.id in read.moment_alias - ): - continue - if everywhere is None: - # Read off the live enum rather than copied out of it, so that a moment - # humanize adds is a moment this rule already knows. Fetched here rather - # than imported with the module: the vocabulary lives beside the drivers. - from hmz.coganchor.agents import EVERYWHERE - - everywhere = frozenset(one.name for one in EVERYWHERE) - if moment.attr in everywhere or moment.attr in declared: - continue - yield Finding( - "unsaid-moment", - "warning", - read.where, - node.lineno, - f"a hook is hung on Moment.{moment.attr}, which only some backends run, and " - "no place declares it -- write Annotated[Agent, Moment." - f"{moment.attr}] where the place is declared, so an agent that cannot run " - "it is refused before its first turn rather than hours in", - ) - - -# --------------------------------------------------------------------------------------- -# The function rules: what is asked of what the flow drives, and how its loops end. -# --------------------------------------------------------------------------------------- - - -class _Crew(NamedTuple): - """A tuple of agents as one function holds it: the places, or only the count. - - Attributes: - fields: One (name, kinds) pair per place for a NamedTuple of them, or None for a - plain tuple, which named nothing. Kinds of None is a place whose annotation this - file cannot read, which is tracked and asked nothing. - kinds: The kind of each element by position, for a plain tuple that said them. - """ - - fields: tuple[tuple[str, frozenset[str] | None], ...] | None - kinds: tuple[frozenset[str] | None, ...] = () - - def held(self) -> frozenset[str] | None: - """What one element of this crew is, where every place is the same thing.""" - each = ( - [kinds for _, kinds in self.fields] - if self.fields is not None - else list(self.kinds) - ) - if not each or any(not one for one in each): - return None - return frozenset(kind for one in each if one for kind in one) - - -class _Answer(NamedTuple): - """One name holding what a turn answered, and how the turn was taken.""" - - shaped: bool - suppressed: bool - line: int - #: The name of the shape the turn was held to, where it was named plainly -- "" for a - #: turn held to no shape, or to one written some way this reading does not follow. - model: str = "" - - -@dataclass -class _Asks: - """What each kind of tracked thing may be asked, read once per checking.""" - - _surfaces: dict[str, frozenset[str]] = field( - default_factory=dict[str, frozenset[str]] - ) - - def allowed(self, kinds: frozenset[str]) -> frozenset[str]: - """Every name something of these kinds answers to.""" - held: frozenset[str] = frozenset() - for kind in kinds: - if kind not in self._surfaces: - self._surfaces[kind] = surface(_KINDS[kind]) - held |= self._surfaces[kind] - return held - - -@dataclass -class _Scope: - """One function being read: what is bound to what, and what was found so far.""" - - read: _Read - asks: _Asks - bindings: dict[str, frozenset[str] | _Crew] = field( - default_factory=dict[str, "frozenset[str] | _Crew"] - ) - answers: dict[str, _Answer] = field(default_factory=dict[str, "_Answer"]) - guarded: set[str] = field(default_factory=set[str]) - #: Attribute reads off a shaped, suppressed answer: (name, line) apiece. - reads: list[tuple[str, int]] = field(default_factory=list[tuple[str, int]]) - #: Whether the function holds a bound of its own -- a spent() call, a range(), an - #: ordering comparison against a number -- which is what excuses its loops. - bounded: bool = False - findings: list[Finding] = field(default_factory=list[Finding]) - - def forgot(self, name: str) -> None: - """Stops tracking one name, which is what any doubtful binding does to it.""" - self.bindings.pop(name, None) - self.answers.pop(name, None) - - -def _functions( - read: _Read, asks: _Asks, skip: frozenset[int] = frozenset() -) -> Iterator[Finding]: - """Reads every function in one file for what it asks and how its loops end. - - Args: - read: The file. - asks: The interface surfaces, shared across the files of one checking. - skip: Functions not to read, by the identity of the `ast` node each is -- the bodies - an atlas compiles, which are declarations rather than programs. - - Yields: - The findings, function by function. - """ - for node in read.tree.body: - yield from _defined(node, read, asks, {}, skip) - - -def _defined( - node: ast.stmt, - read: _Read, - asks: _Asks, - inherited: dict[str, frozenset[str] | _Crew], - skip: frozenset[int] = frozenset(), -) -> Iterator[Finding]: - """One top-level statement, read for the functions in it.""" - if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)): - if id(node) not in skip: - yield from _function(node, read, asks, inherited) - elif isinstance(node, ast.ClassDef): - for held in node.body: - yield from _defined(held, read, asks, inherited, skip) - elif isinstance(node, ast.If): - for held in [*node.body, *node.orelse]: - yield from _defined(held, read, asks, inherited, skip) - - -def _function( - node: ast.FunctionDef | ast.AsyncFunctionDef, - read: _Read, - asks: _Asks, - inherited: dict[str, frozenset[str] | _Crew], -) -> Iterator[Finding]: - """One function: bindings followed in order, then its loops read against them. - - Args: - node: The function. - read: Its file. - asks: The interface surfaces. - inherited: What the enclosing function had bound, which a closure reads. - """ - scope = _Scope(read, asks, bindings=dict(inherited)) - args = node.args - params = [*args.posonlyargs, *args.args, *args.kwonlyargs] - for param in params: - scope.forgot(param.arg) - kind = _annotated(param.annotation, read.proto) - crew = _crewed(param.annotation, read) if not kind else None - if kind: - scope.bindings[param.arg] = kind - elif crew is not None: - scope.bindings[param.arg] = crew - for one in (args.vararg, args.kwarg): - if one is not None: - scope.forgot(one.arg) - _statements(node.body, scope, read, asks) - yield from scope.findings - for name, line in scope.reads: - if name not in scope.guarded: - yield Finding( - "unguarded-answer", - "warning", - read.where, - line, - f"{name} may be None here -- a suppressed turn held to a shape answers " - "with nothing when it fails, and a field read off nothing ends the run " - "where a guard would have taken the turn again", - ) - if not _yields(node): - yield from _loops(node, scope) - - -def _annotated( - annotation: ast.expr | None, proto: Mapping[str, str] -) -> frozenset[str] | None: - """What kinds of driven thing one annotation says a name is, or None for no answer. - - Asked of the interface names rather than of a whole file, so that the compiling next - door -- which gathers those names across every file a flow holds -- asks this one - reading rather than a looser one of its own. - - Args: - annotation: The annotation. - proto: The local name of each flow-facing interface, as this flow imports them. - - Returns: - One name per interface it says the thing is, and None where it says nothing. - """ - said = _unquoted(annotation) - if said is None: - return None - if isinstance(said, ast.Subscript) and _root(said.value) == "Annotated": - first = next(iter(_elements(said.slice)), None) - return _annotated(first, proto) - if isinstance(said, ast.Subscript) and _root(said.value) == "Optional": - first = next(iter(_elements(said.slice)), None) - return _annotated(first, proto) - if isinstance(said, ast.BinOp) and isinstance(said.op, ast.BitOr): - left = _annotated(said.left, proto) - right = _annotated(said.right, proto) - if left is None and right is None: - return None - # `Agent | None` is an agent to be guarded; `Agent | Session` is either, and is - # asked only what both answer to being too strict -- so it is asked the union. - return (left or frozenset()) | (right or frozenset()) - if isinstance(said, ast.Constant) and said.value is None: - return frozenset() - if isinstance(said, ast.Name) and said.id in proto: - return frozenset({proto[said.id]}) - return None - - -def _crewed(annotation: ast.expr | None, read: _Read) -> _Crew | None: - """The tuple of agents one annotation declares, where this file can read it. - - A crew it is not sure of is no crew at all: a NamedTuple with one place this file - cannot read as an interface is somebody's data rather than the agents, and tracking - it would find askings that are nobody's business here. - """ - said = _unquoted(annotation) - if said is None: - return None - if isinstance(said, ast.Name) and said.id in read.crews: - fields = [ - (node.target.id, _annotated(node.annotation, read.proto)) - for node in read.crews[said.id].body - if isinstance(node, ast.AnnAssign) and isinstance(node.target, ast.Name) - ] - if fields and all(kinds for _, kinds in fields): - return _Crew(tuple(fields)) - return None - if isinstance(said, ast.Subscript) and _root(said.value) == "tuple": - kinds = tuple( - _annotated(one, read.proto) - for one in _elements(said.slice) - if not (isinstance(one, ast.Constant) and one.value is Ellipsis) - ) - if kinds and all(kinds): - return _Crew(None, kinds) - return None - - -def _unquoted(annotation: ast.expr | None) -> ast.expr | None: - """One annotation with any quoting read through, since a string is still the words.""" - if isinstance(annotation, ast.Constant) and isinstance(annotation.value, str): - try: - return ast.parse(annotation.value, mode="eval").body - except (SyntaxError, ValueError): - return None - return annotation - - -def _elements(slice_: ast.expr) -> tuple[ast.expr, ...]: - """The elements of one subscript, one or many.""" - return tuple(slice_.elts) if isinstance(slice_, ast.Tuple) else (slice_,) - - -def _names_in(annotation: ast.expr) -> set[str]: - """Every plain name one annotation mentions, quoting and all.""" - said = _unquoted(annotation) - if said is None: - return set() - return {node.id for node in ast.walk(said) if isinstance(node, ast.Name)} - - -def _statements(body: list[ast.stmt], scope: _Scope, read: _Read, asks: _Asks) -> None: - """Walks statements in order, checking what they ask and following what they bind.""" - for node in body: - if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)): - scope.findings.extend(_function(node, read, asks, dict(scope.bindings))) - scope.forgot(node.name) - elif isinstance(node, ast.ClassDef): - for held in node.body: - if isinstance(held, (ast.FunctionDef, ast.AsyncFunctionDef)): - scope.findings.extend(_function(held, read, asks, {})) - scope.forgot(node.name) - elif isinstance(node, ast.Assign): - _expression(node.value, scope) - _assigned(node.targets, node.value, scope) - elif isinstance(node, ast.AnnAssign): - if node.value is not None: - _expression(node.value, scope) - if isinstance(node.target, ast.Name): - scope.forgot(node.target.id) - kind = _annotated(node.annotation, scope.read.proto) - if kind: - scope.bindings[node.target.id] = kind - elif node.value is not None: - _assigned([node.target], node.value, scope) - else: - _expression(node.target, scope) - elif isinstance(node, ast.AugAssign): - _expression(node.value, scope) - if isinstance(node.target, ast.Name): - scope.forgot(node.target.id) - elif isinstance(node, (ast.For, ast.AsyncFor)): - _expression(node.iter, scope) - _bound_target(node.target, _element_of(node.iter, scope), scope) - _statements(node.body, scope, read, asks) - _statements(node.orelse, scope, read, asks) - elif isinstance(node, (ast.While, ast.If)): - _guards(node.test, scope) - _expression(node.test, scope) - _statements(node.body, scope, read, asks) - _statements(node.orelse, scope, read, asks) - elif isinstance(node, (ast.With, ast.AsyncWith)): - for item in node.items: - _expression(item.context_expr, scope) - if item.optional_vars is not None: - _bound_target(item.optional_vars, None, scope) - _statements(node.body, scope, read, asks) - elif isinstance(node, ast.Try): - _statements(node.body, scope, read, asks) - for handler in node.handlers: - if handler.name: - scope.forgot(handler.name) - _statements(handler.body, scope, read, asks) - _statements(node.orelse, scope, read, asks) - _statements(node.finalbody, scope, read, asks) - elif isinstance(node, ast.Assert): - _guards(node.test, scope) - _expression(node.test, scope) - elif isinstance(node, ast.Delete): - for target in node.targets: - if isinstance(target, ast.Name): - scope.forgot(target.id) - elif isinstance(node, (ast.Import, ast.ImportFrom)): - for alias in node.names: - scope.forgot((alias.asname or alias.name).split(".")[0]) - elif isinstance(node, (ast.Return, ast.Expr)): - if node.value is not None: - _expression(node.value, scope) - elif isinstance(node, ast.Raise): - for held in (node.exc, node.cause): - if held is not None: - _expression(held, scope) - elif isinstance(node, ast.Match): - _expression(node.subject, scope) - for case in node.cases: - for name in _captured(case.pattern): - scope.forgot(name) - if case.guard is not None: - _expression(case.guard, scope) - _statements(case.body, scope, read, asks) - - -def _captured(pattern: ast.pattern) -> set[str]: - """Every name one match pattern binds, all of which stop being tracked.""" - said: set[str] = set() - for node in ast.walk(pattern): - if isinstance(node, (ast.MatchAs, ast.MatchStar)) and node.name: - said.add(node.name) - elif isinstance(node, ast.MatchMapping) and node.rest: - said.add(node.rest) - return said - - -def _assigned(targets: list[ast.expr], value: ast.expr, scope: _Scope) -> None: - """Follows one assignment: what the targets are now bound to, if anything tracked.""" - held = _valued(value, scope) - for target in targets: - if isinstance(target, ast.Name): - scope.forgot(target.id) - if isinstance(held, _Answer): - scope.answers[target.id] = held - elif held is not None: - scope.bindings[target.id] = held - elif isinstance(target, (ast.Tuple, ast.List)): - _unpacked(target, value, scope) - else: - _expression(target, scope) - - -def _unpacked(target: ast.Tuple | ast.List, value: ast.expr, scope: _Scope) -> None: - """Follows a tuple unpacking, which is how `(agent,) = agents` hands a place over.""" - crew = scope.bindings.get(value.id) if isinstance(value, ast.Name) else None - each = crew.held() if isinstance(crew, _Crew) else None - kinds: list[frozenset[str] | None] | None = None - if isinstance(crew, _Crew): - held = ( - [one for _, one in crew.fields] - if crew.fields is not None - else list(crew.kinds) - ) - kinds = held if len(held) == len(target.elts) else None - for at, one in enumerate(target.elts): - if isinstance(one, ast.Name): - scope.forgot(one.id) - bound = kinds[at] if kinds is not None else each - if isinstance(crew, _Crew) and bound: - scope.bindings[one.id] = bound - elif isinstance(one, (ast.Tuple, ast.List)): - _unpacked(one, value, scope) - elif isinstance(one, ast.Starred) and isinstance(one.value, ast.Name): - scope.forgot(one.value.id) - - -def _bound_target(target: ast.expr, kind: frozenset[str] | None, scope: _Scope) -> None: - """Binds a loop or with target: to the element kind where there is one, else to doubt.""" - if isinstance(target, ast.Name): - scope.forgot(target.id) - if kind: - scope.bindings[target.id] = kind - elif isinstance(target, (ast.Tuple, ast.List)): - for one in target.elts: - _bound_target(one, None, scope) - - -def _element_of(iterable: ast.expr, scope: _Scope) -> frozenset[str] | None: - """What iterating one expression yields, where it is a tracked crew.""" - if isinstance(iterable, ast.Name): - held = scope.bindings.get(iterable.id) - if isinstance(held, _Crew): - return held.held() - return None - - -def _valued(value: ast.expr, scope: _Scope) -> frozenset[str] | _Crew | _Answer | None: - """What one expression is, as far as the bindings can say. - - Args: - value: The expression on the right of an assignment. - scope: The function so far. - - Returns: - A kind for something driven, a crew for a tuple of them, an answer for what a turn - of one said, and None for anything this reading does not follow -- which stops the - target being tracked rather than mistracking it. - """ - if isinstance(value, ast.Name): - return scope.bindings.get(value.id) - if isinstance(value, ast.Await): - return _valued(value.value, scope) - if isinstance(value, ast.Attribute): - held = ( - scope.bindings.get(value.value.id) - if isinstance(value.value, ast.Name) - else None - ) - if isinstance(held, _Crew) and held.fields is not None: - return next( - (kinds for name, kinds in held.fields if name == value.attr), None - ) - return None - if isinstance(value, ast.Subscript): - held = ( - scope.bindings.get(value.value.id) - if isinstance(value.value, ast.Name) - else None - ) - return held.held() if isinstance(held, _Crew) else None - if isinstance(value, ast.Call): - return _called(value, scope) - return None - - -def _called(value: ast.Call, scope: _Scope) -> frozenset[str] | _Crew | _Answer | None: - """What calling one thing answers with, where what is called is tracked.""" - func = value.func - opened = isinstance(func, ast.Attribute) and func.attr in {"new", "clone"} - target = func.value if isinstance(func, ast.Attribute) and opened else func - held = _valued(target, scope) if opened else None - if isinstance(func, ast.Attribute) and func.attr == "new": - if isinstance(held, frozenset) and held: - return frozenset({"session"}) - return None - if isinstance(func, ast.Attribute) and func.attr == "clone": - return held if isinstance(held, frozenset) else None - spoken = _valued(func, scope) - if isinstance(spoken, frozenset) and spoken: - # Calling an agent or a session is a turn, and what it answers is an answer. - return _Answer( - shaped=any(one.arg == "schema" for one in value.keywords), - suppressed=any( - one.arg == "suppress" - and isinstance(one.value, ast.Constant) - and one.value.value is True - for one in value.keywords - ), - line=value.lineno, - model=_shaped_as(value), - ) - if isinstance(func, ast.Attribute) and func.attr in {"aturn", "pursue", "apursue"}: - asked = _valued(func.value, scope) - if isinstance(asked, frozenset) and asked: - return _Answer( - shaped=any(one.arg == "schema" for one in value.keywords), - suppressed=any( - one.arg == "suppress" - and isinstance(one.value, ast.Constant) - and one.value.value is True - for one in value.keywords - ), - line=value.lineno, - model=_shaped_as(value), - ) - return None - - -def _shaped_as(value: ast.Call) -> str: - """The name of the shape one turn is held to, where the call names it plainly.""" - for keyword in value.keywords: - if keyword.arg == "schema" and isinstance(keyword.value, ast.Name): - return keyword.value.id - return "" - - -#: The comparisons that read as a bound: a count against something, ordered. -_ORDERED = (ast.Lt, ast.LtE, ast.Gt, ast.GtE) - - -def _expression(node: ast.expr, scope: _Scope) -> None: - """Checks one expression tree against the bindings, and notes what it holds. - - What is checked is every attribute asked of a tracked name; what is noted is every - guard on an answer, and whether the function holds a bound of its own. - """ - if isinstance(node, ast.Attribute): - _asked(node, scope) - _expression(node.value, scope) - return - if isinstance(node, ast.Call): - if isinstance(node.func, ast.Attribute) and node.func.attr == "spent": - scope.bounded = True - if isinstance(node.func, ast.Name) and node.func.id == "range": - scope.bounded = True - for one in [node.func, *node.args]: - _expression(one, scope) - for kw in node.keywords: - _expression(kw.value, scope) - return - if isinstance(node, ast.Compare): - if any(isinstance(op, _ORDERED) for op in node.ops) and any( - isinstance(side, ast.Constant) - and isinstance(side.value, (int, float)) - and not isinstance(side.value, bool) - for side in [node.left, *node.comparators] - ): - scope.bounded = True - _never(node, scope) - _guards(node, scope) - for one in [node.left, *node.comparators]: - _expression(one, scope) - return - if isinstance(node, ast.BoolOp): - _guards(node, scope) - for one in node.values: - _expression(one, scope) - return - if isinstance(node, ast.IfExp): - _guards(node.test, scope) - for one in (node.test, node.body, node.orelse): - _expression(one, scope) - return - if isinstance(node, ast.NamedExpr): - _expression(node.value, scope) - scope.forgot(node.target.id) - held = _valued(node.value, scope) - if isinstance(held, _Answer): - scope.answers[node.target.id] = held - elif held is not None: - scope.bindings[node.target.id] = held - return - if isinstance(node, ast.Lambda): - return # its own scope, and nothing in it runs here - if isinstance(node, (ast.ListComp, ast.SetComp, ast.GeneratorExp, ast.DictComp)): - _comprehended(node, scope) - return - for one in ast.iter_child_nodes(node): - if isinstance(one, ast.expr): - _expression(one, scope) - - -def _comprehended( - node: ast.ListComp | ast.SetComp | ast.GeneratorExp | ast.DictComp, - scope: _Scope, -) -> None: - """One comprehension, read with its own targets bound for as long as it lasts.""" - before = dict(scope.bindings) - for gen in node.generators: - _expression(gen.iter, scope) - _bound_target(gen.target, _element_of(gen.iter, scope), scope) - for test in gen.ifs: - _guards(test, scope) - _expression(test, scope) - if isinstance(node, ast.DictComp): - _expression(node.key, scope) - _expression(node.value, scope) - else: - _expression(node.elt, scope) - scope.bindings = before - - -def _guards(test: ast.expr, scope: _Scope) -> None: - """Notes every answer one test stands guard over. - - A guard is the answer read as a truth -- `if worked:` -- or compared against None, - either way round. Noted wherever it appears rather than matched to a branch: what the - warning is for is the answer nobody tested at all. - """ - # Through the walrus, which is how the answer and the guard over it are written on one - # line -- `if (said := agent(..., schema=X)) is not None:` -- and reading it as work - # rather than as a name would be a warning about a flow that guarded exactly right. - test = _named(test) - if isinstance(test, ast.Name): - scope.guarded.add(test.id) - elif isinstance(test, ast.UnaryOp): - held = _named(test.operand) - if isinstance(held, ast.Name): - scope.guarded.add(held.id) - elif isinstance(test, ast.BoolOp): - for one in test.values: - _guards(one, scope) - elif isinstance(test, ast.Compare): - sides = [_named(one) for one in (test.left, *test.comparators)] - against_none = any( - isinstance(side, ast.Constant) and side.value is None for side in sides - ) - if against_none: - for side in sides: - if isinstance(side, ast.Name): - scope.guarded.add(side.id) - - -def _named(node: ast.expr) -> ast.expr: - """One test with the walrus around it taken off, which is the name it binds. - - Args: - node: The test, or one side of it. - - Returns: - The name a `:=` binds, and the node itself where there is no `:=` -- so that an answer - bound and tested on one line reads as the name it was bound to. - """ - return node.target if isinstance(node, ast.NamedExpr) else node - - -def _asked(node: ast.Attribute, scope: _Scope) -> None: - """One attribute asked of a name, checked where the name is tracked.""" - if not isinstance(node.value, ast.Name) or node.attr.startswith("_"): - return - name = node.value.id - answer = scope.answers.get(name) - if answer is not None and answer.shaped and answer.suppressed: - scope.reads.append((name, node.lineno)) - return - held = scope.bindings.get(name) - if held is None: - return - if isinstance(held, _Crew): - places: frozenset[str] = ( - frozenset(field for field, _ in held.fields) - if held.fields is not None - else frozenset() - ) - if node.attr in places | _OF_A_TUPLE: - return - which = ( - ", ".join(field for field, _ in held.fields) - if held.fields - else "how many there are, and nothing else" - ) - scope.findings.append( - Finding( - "unknown-ask", - "error", - scope.read.where, - node.lineno, - f"the agents have no place called {node.attr!r} -- the flow declares " - f"{which}, and a place it does not declare is not one it was handed", - ) - ) - return - if node.attr in scope.asks.allowed(held): - return - what = " or ".join( - f"hmz._legacy_flows.{kind.capitalize()}" for kind in sorted(held) - ) - scope.findings.append( - Finding( - "unknown-ask", - "error", - scope.read.where, - node.lineno, - f"nothing here answers to {node.attr!r} -- what a flow may ask is written " - f"on {what}, and a name that is not there fails at the first turn", - ) - ) - - -#: The comparisons that ask whether a value is exactly one of some others. -_BY_VALUE = (ast.Eq, ast.NotEq, ast.In, ast.NotIn) - - -def _never(node: ast.Compare, scope: _Scope) -> None: - """One comparison read against the shape behind it, for a value no answer holds. - - `review.verdict == "DONE"` over `Literal["done", "redo"]` is a guard that never opens, - and `!= "DONE"` one that never shuts: either way the flow steers by a value the shape - cannot answer with. Said only where everything is certain -- the answer's shape is a - model this same file declares, the field spells its values out, and the other side is - constants -- so a shape read from elsewhere is let be rather than guessed at. - """ - if len(node.ops) != 1 or not isinstance(node.ops[0], _BY_VALUE): - return - membership = isinstance(node.ops[0], (ast.In, ast.NotIn)) - pairs = [(node.left, node.comparators[0])] - if not membership: - # `"done" == review.verdict` reads the same either way round; `"d" in x` does not. - pairs.append((node.comparators[0], node.left)) - for asked, against in pairs: - if not (isinstance(asked, ast.Attribute) and isinstance(asked.value, ast.Name)): - continue - answer = scope.answers.get(asked.value.id) - if answer is None or not answer.model: - continue - held = _offered(answer.model, scope.read, asked.attr, set()) - values = _values_of(against, membership=membership) - if held is None or values is None: - return - offers = ", ".join(repr(one) for one in sorted(held, key=repr)) - for value in values: - if value not in held: - scope.findings.append( - Finding( - "unknown-verdict", - "warning", - scope.read.where, - node.lineno, - f"no answer holds {value!r} at {asked.attr!r} -- the shape " - f"offers {offers}, and a comparison against a value it cannot " - "hold reads as a guard and guards nothing", - ) - ) - return - - -def _offered( - model: str, read: _Read, field_name: str, seen: set[str] -) -> frozenset[object] | None: - """Every value one field of a model may hold, read off the model's own words. - - Follows local bases the way the strictness rule does -- a field a model inherits is as - much its shape as one it declares. - - Args: - model: The model's name. - read: The file, whose models are the only ones read. - field_name: The field asked about. - seen: The models already walked, which stops a circular inheritance. - - Returns: - The values, or None where the model or the field is not here, or the field's - annotation does not spell its values out. - """ - if model in seen or model not in read.models: - return None - seen.add(model) - declared = read.models[model] - for node in declared.body: - if ( - isinstance(node, ast.AnnAssign) - and isinstance(node.target, ast.Name) - and node.target.id == field_name - ): - return _options_in(node.annotation) - for base in declared.bases: - if isinstance(base, ast.Name): - held = _offered(base.id, read, field_name, seen) - if held is not None: - return held - return None - - -def _options_in(annotation: ast.expr | None) -> frozenset[object] | None: - """Every value one annotation admits, where it spells them all out. - - `Literal` through and through -- unions, `Optional` and `Annotated` read through -- - or None for anything open: a field that may hold a plain `str` is a field any - comparison against is an honest one. - """ - said = _unquoted(annotation) - if said is None: - return None - if isinstance(said, ast.Constant) and said.value is None: - return frozenset({None}) - if isinstance(said, ast.BinOp) and isinstance(said.op, ast.BitOr): - left = _options_in(said.left) - right = _options_in(said.right) - if left is None or right is None: - return None - return left | right - if not isinstance(said, ast.Subscript): - return None - head = _tip(said.value) - parts = _elements(said.slice) - if head == "Literal": - options: set[object] = set() - for part in parts: - if not isinstance(part, ast.Constant): - return None - options.add(part.value) - return frozenset(options) - if head == "Annotated" and parts: - return _options_in(parts[0]) - if head == "Optional" and parts: - inner = _options_in(parts[0]) - return None if inner is None else inner | {None} - if head == "Union": - gathered: frozenset[object] = frozenset() - for part in parts: - inner = _options_in(part) - if inner is None: - return None - gathered |= inner - return gathered - return None - - -def _values_of(node: ast.expr, *, membership: bool) -> list[object] | None: - """The constant values one side of a comparison holds, or None where it is not sure.""" - if not membership: - return [node.value] if isinstance(node, ast.Constant) else None - if isinstance(node, (ast.Tuple, ast.List, ast.Set)): - values: list[object] = [] - for elt in node.elts: - if not isinstance(elt, ast.Constant): - return None - values.append(elt.value) - return values - return None - - -# --------------------------------------------------------------------------------------- -# The loop rules: a loop is legal when something inside it can end it. -# --------------------------------------------------------------------------------------- - - -def _yields(node: ast.FunctionDef | ast.AsyncFunctionDef) -> bool: - """Whether one function is a generator, whose loops end where their consumer stops.""" - waiting: list[ast.AST] = list(ast.iter_child_nodes(node)) - while waiting: - held = waiting.pop() - if isinstance(held, (ast.FunctionDef, ast.AsyncFunctionDef, ast.Lambda)): - continue - if isinstance(held, (ast.Yield, ast.YieldFrom)): - return True - waiting.extend(ast.iter_child_nodes(held)) - return False - - -class _Exit(NamedTuple): - """One way out of a loop, and the conditions standing between the loop and it.""" - - line: int - conditions: tuple[ast.expr, ...] - - -def _loops( - node: ast.FunctionDef | ast.AsyncFunctionDef, scope: _Scope -) -> Iterator[Finding]: - """Every constant-true loop in one function, read for how it ends. - - Args: - node: The function, with its bindings already followed. - scope: What following them collected. - - Yields: - A `dead-loop` error for one nothing inside can end and no turn is taken in, a - `sleeping-loop` error for one that only sleeps -- alive from the outside and doing - nothing -- and an `unbounded-loop` warning for one that ends only when the run's - allowance is spent or when an agent says so, in a function with no bound of its own. - """ - for loop in _whiles(node.body): - if not (isinstance(loop.test, ast.Constant) and loop.test.value): - continue - exits = _exits(loop.body, ()) - turns = _turned(loop.body, scope) - if not exits and not turns: - if _sleeps(loop.body): - code = "sleeping-loop" - said = ( - "this loop only sleeps -- from outside it looks alive, and each " - "round does nothing; a loop earns its keep by doing something " - "that can end it" - ) - else: - code = "dead-loop" - said = ( - "this loop cannot end -- no break, no return, no raise inside it, " - "and no turn of an agent to run out of the run's allowance; a loop " - "is legal when something inside it can end it" - ) - yield Finding(code, "error", scope.read.where, loop.lineno, said) - continue - if scope.bounded: - continue - shaped = {name for name, answer in scope.answers.items() if answer.shaped} - if not exits: - # A loop with no way out but the money running out. Legal, because the run's - # allowance ends it wherever it is -- and worth saying, because how long that - # takes is what somebody set in the menu rather than anything this flow decides. - yield Finding( - "unbounded-loop", - "warning", - scope.read.where, - loop.lineno, - "nothing inside this loop ends it, so it runs until the run's allowance " - "is spent -- which is a stop and not a finish; give it a bound of its own " - "if it is meant to end on having done something", - ) - elif all(_by_verdict(one, shaped) for one in exits): - yield Finding( - "unbounded-loop", - "warning", - scope.read.where, - loop.lineno, - "every way out of this loop waits for an agent to say so, and an agent " - "may never say it -- it then runs to the end of the run's allowance; " - "give the loop a bound of its own if it should stop sooner: a cap on the " - "rounds, a range, a clock", - ) - - -def _turned(body: list[ast.stmt], scope: _Scope) -> bool: - """Whether anything in one loop takes a turn of an agent. - - Which is what makes a loop with no `break` in it legal: every session of every backend - is held to the run's allowance, and a turn taken under a spent one raises `Stopped` - rather than answering. So a loop of turns ends, wherever it is, and a flow that wrote a - budget check of its own to have an exit would be writing the one thing a flow must not. - - Args: - body: The loop's statements. - scope: The function so far, which is what says which names are agents. - - Returns: - Whether a turn is taken in it. - """ - return any( - isinstance(held, ast.Call) and isinstance(_called(held, scope), _Answer) - for one in body - for held in ast.walk(one) - ) - - -def _whiles(body: list[ast.stmt]) -> Iterator[ast.While]: - """Every while loop in one function's own body, nested functions left to themselves.""" - for node in body: - if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)): - continue - if isinstance(node, ast.While): - yield node - for held in _blocks(node): - yield from _whiles(held) - - -def _blocks(node: ast.stmt) -> Iterator[list[ast.stmt]]: - """The statement blocks one statement holds, whichever shape it is.""" - for name in ("body", "orelse", "finalbody"): - held = getattr(node, name, None) - if isinstance(held, list) and held and isinstance(held[0], ast.stmt): - yield held - for handler in getattr(node, "handlers", []): - yield handler.body - for case in getattr(node, "cases", []): - yield case.body - - -def _exits(body: list[ast.stmt], conditions: tuple[ast.expr, ...]) -> list[_Exit]: - """Every way out of a loop with this body, each with the conditions guarding it. - - A `break` of the loop's own, a `return`, a `raise`: anything that ends the loop or the - function around it. A `break` inside a nested loop ends that loop instead, and is not - one; a `return` inside one still ends the function, and is. - """ - found: list[_Exit] = [] - for node in body: - if isinstance(node, (ast.Break, ast.Return, ast.Raise)): - found.append(_Exit(node.lineno, conditions)) - elif isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)): - continue - elif isinstance(node, ast.If): - found.extend(_exits(node.body, (*conditions, node.test))) - found.extend(_exits(node.orelse, (*conditions, node.test))) - elif isinstance(node, (ast.While, ast.For, ast.AsyncFor)): - # A break in there is that loop's; a return or a raise is still a way out. - for held in _blocks(node): - found.extend( - one - for one in _exits(held, conditions) - if not _breaks_at(node, one.line) - ) - else: - for held in _blocks(node): - found.extend(_exits(held, conditions)) - return found - - -def _breaks_at(loop: ast.stmt, line: int) -> bool: - """Whether the exit at this line is a break belonging to this nested loop.""" - return any( - isinstance(node, ast.Break) and node.lineno == line - for node in ast.walk(loop) - if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)) - ) - - -def _by_verdict(exit_: _Exit, shaped: set[str]) -> bool: - """Whether one way out waits on a field of what an agent answered. - - The `review.done` shape: a condition reading an attribute off an answer the turn was - held to. A turn merely having landed -- `if said:` -- is not one, since a loop that - takes a failed turn again is bounded by the turns landing, not by what they say. - """ - for condition in exit_.conditions: - for node in ast.walk(condition): - if ( - isinstance(node, ast.Attribute) - and isinstance(node.value, ast.Name) - and node.value.id in shaped - and not node.attr.startswith(("_", "model_")) - ): - return True - return False - - -def _sleeps(body: list[ast.stmt]) -> bool: - """Whether a loop body does nothing but wait: sleeps, passes, and constants.""" - for node in body: - if isinstance(node, (ast.Pass, ast.Continue)): - continue - if isinstance(node, ast.Expr): - value = node.value - if isinstance(value, ast.Constant): - continue - if isinstance(value, ast.Call) and ( - (isinstance(value.func, ast.Name) and value.func.id == "sleep") - or ( - isinstance(value.func, ast.Attribute) and value.func.attr == "sleep" - ) - ): - continue - return False - if isinstance(node, ast.Assign) and isinstance(node.value, ast.Constant): - continue - return False - return True - - -# --------------------------------------------------------------------------------------- -# The capability catalogue: what this installed humanize serves, read off it live. -# --------------------------------------------------------------------------------------- - - -#: The two halves of :class:`hmz.coganchor.agents.config.Needs`, by the name of the field a -#: flow writes each under. Every capability belongs to exactly one of them, and which one is -#: not a matter of taste: a name in the first is answered by the backend filling the place and -#: a name in the second by the machine its turns land on, and the two are read by different -#: code against different facts. A capability that did not say which half it was asked in is -#: one a flow can write in the wrong one and never be told -- `Needs("isolated")` was exactly -#: that, silently satisfied by every backend there is, because a machine capability carries no -#: backends and an empty backend set means every backend. -OF_AGENT = "of_agent" -WHERE = "where" - - -class Capability(NamedTuple): - """One thing a flow may build on, and which backends serve it. - - Attributes: - name: What it is called: a primitive by one word, and a moment only some backends - reach as `moment:`. - backends: The backends that serve it, or empty for one every backend serves. Meaningful - only where `asked` is :data:`OF_AGENT`: what a machine comes to is no backend's to - answer, so a `where` capability carries none and must not be read as carrying all. - said: What the ask looks like, with the code spelled out: this is what a compiler - or a person choosing what to build on is shown. - asked: Which half of :class:`~hmz.coganchor.agents.config.Needs` asks for it -- - :data:`OF_AGENT` or :data:`WHERE`. Most are asked of the agent, so that is the - default and the ones that are not say so. - """ - - name: str - backends: frozenset[str] - said: str - asked: str = OF_AGENT - - -#: Where an agent's turns may land, and what a flow writes to say so. Asked of the machine -#: rather than of the agent: what a machine is is the same question whichever CLI is being -#: driven there, and what answers is -#: :attr:`~hmz.coganchor.machines.MachineConfig.capabilities`. -#: -#: The words are :mod:`hmz.coganchor.places`' and only the prose is this module's, which is the -#: line this file draws for the whole vocabulary: a capability name is a fact somebody -#: declares and an ask a flow writes, the fact belongs wherever the thing it is about is -#: written down, and the ask belongs here. Keyed off the constants rather than off strings, so -#: that a place humanize learns to reach cannot be described here under a name nothing -#: answers to -- :func:`_places` checks the two sets against each other. -_PLACES = { - places.REMOTE: "an agent whose turns land on another machine -- Annotated[Agent, Remote] -- " - "and only a place that says so may be given one at all; the agent goes on running here, " - "and what moves is the project it reads and the commands it runs", - places.ISOLATED: "an agent that works in a container of its own -- " - "Annotated[Agent, Isolated('python:3.12')] -- which the flow names and nobody may " - "configure: what is isolated is the tools a command finds and not the work", - places.MANAGED: "the machine brought up for the agent and taken down after it, as against an " - "anchor onto one that was already running -- which is never stopped here, a machine " - "nobody here started being nobody here's to end", - "linux": "an agent's turns landing on a Linux machine, which is what a container started " - "here is and where a turn's commands are reached by tracing them", - "darwin": "an agent's turns landing on a macOS machine, which is one that is already up " - "and reached by an anchor onto it: nothing here starts one", -} - -#: The anchors nothing has to be written down for, each true of every CLI here: every one of -#: them is a command line humanize spawns, and what a spawned turn runs is what a supervisor -#: traces. The rest are read off the backends' own facts, under the same name. -_REACHED = places.ROADS - -#: The reach names humanize can only serve where the turn runs on this machine, re-exported -#: from :mod:`hmz.coganchor.places` because :func:`hmz.runtime.flowing.driving.serves` is what takes -#: them away again and has always reached for them here. -INSIDE = places.INSIDE - -#: How a turn's own commands are reached, which is what an anchor is made of, by the name a -#: flow and a compiler ask for each under -- and what the ask looks like. -#: -#: Split across the two halves of `Needs`, which is the one thing about this vocabulary that -#: is not obvious and used to be written down wrong. :data:`_REACHED` is asked under `where=`: -#: a road is how an anchor reaches a machine, and -#: :attr:`~hmz.coganchor.anchor.AnchorConfig.capabilities` is what declares one, so it is the -#: machine that was pointed down it which answers. :data:`INSIDE` is asked of the *agent*: a -#: hook table and a preload are the CLI's own to take, :meth:`hmz.coganchor.backends.Profile.tags` -#: is what declares them, and no machine's settings have ever carried either -- a flow writing -#: `Needs(where=("anchor:hooked",))` was refused by every machine there is, which is a check -#: nothing could pass. What remains true of both, and is why they were ever written down -#: together, is that a turn landing somewhere else is a turn humanize cannot reach into: that -#: is not a machine declaring the road, it is the agent losing it, and `serves` is where it is -#: taken away. -_ANCHORS = { - places.NATIVE_CLI: "an anchored turn taken as the CLI already installed on the target, " - "read off its own three streams, with nothing traced and nothing mirrored -- asked of a " - "machine, so a place whose agent was pointed nowhere comes to it no more than to remote", - places.SUPERVISED: "an anchored turn whose commands are reached by tracing the process it " - "runs them in, which is how the work of a turn taken here lands on the machine the flow " - "chose -- asked of a machine, the same way", - places.AFAR: "an anchored turn whose harness -- the agent process and the supervisor " - "tracing it -- runs on a machine other than this one, beside its work or introduced to " - "it through humanize; asked of a machine, the same way", - places.HOOKED: "a turn reached through the CLI's own hooks, written for this run and " - "read by no other -- backends.named().hooks says through which seam it is told; " - "asked of the agent, and taken away from one whose turns land on another machine", - places.PRELOADED: "a turn reached from inside the process, by what its runtime is told " - "to load before it starts -- backends.named().preloads names the variable that " - "carries it; asked of the agent, and taken away from one whose turns land elsewhere", -} - - -#: The facts about a CLI that are a capability of their own, by the name -#: :meth:`hmz.coganchor.backends.Profile.tags` already answers with -- and what the ask looks -#: like. `fork` is not here because it has a sentence of its own beside the session -#: capabilities it belongs with, and the two `anchor:` tags are in :data:`_ANCHORS` for the -#: same reason. -#: -#: They were askable before they were catalogued, which is the bug this closes: -#: :func:`hmz.runtime.flowing.driving.comes_to` unions a backend's tags in, so `Needs("search")` has -#: always worked, while the catalogue and :func:`briefed` -- the one page a compiler steers -#: by -- said nothing about them. A name a flow may write and a compiler cannot read is a -#: name a generated flow will never ask under. -_PROFILED = { - "search": "the CLI's own web search, which is the thing AgentConfig(web_search=False) " - "takes away -- a backend not among these has none to take away, and is refused where the " - "agent is made rather than quietly reaching the web anyway", - "swarm": "the CLI runs one turn as several agents at once, which its own ladder says by " - "carrying a `swarm`-prefixed rung -- AgentConfig(effort='swarmmax') on a backend among " - "these, and the prefix comes off before the rung is read", - "resume": "one conversation picked back up across turns, which is what a session here is " - "made of -- a backend not among these opens a new conversation per turn, so a flow that " - "holds a session across turns holds nothing", -} - - -def _tagged(tag: str, agents: Mapping[str, type]) -> frozenset[str]: - """Which of these backends the facts written down about them say serve one capability. - - Args: - tag: The capability name, as :meth:`hmz.coganchor.backends.Profile.tags` spells it. - agents: The backends there are, by name. - - Returns: - The names that serve it. A driven backend with no profile at all serves none of them: - what is not written down is not a fact about that CLI. - """ - from hmz.coganchor.backends import named - - return frozenset( - name - for name in agents - if (profile := named(name)) is not None and tag in profile.tags() - ) - - -def _places() -> list[Capability]: - """Where an agent's turns may land, as a flow says it. - - Returns: - One capability apiece, each asked under `where=` and carrying no backends: a machine is - no backend's to answer for, whichever CLI is being driven on it. - - Raises: - RuntimeError: If coganchor knows a place this module has no sentence for, or has a - sentence here for one nothing answers to. The words are declared there and described - here, and a description with nothing behind it is a capability a flow can ask for and - never be given. - """ - described = frozenset(_PLACES) - if described != places.PLACES: - raise RuntimeError( - f"the places described here and the places there are differ: " - f"{sorted(described ^ places.PLACES)}" - ) - return [ - Capability(name, frozenset(), said, WHERE) for name, said in _PLACES.items() - ] - - -def _anchors(agents: Mapping[str, type]) -> list[Capability]: - """How a turn's own commands are reached here, read off what serves each way. - - An anchor nothing here serves yet is left out rather than listed with nobody against it: - an empty backend set means every backend in this catalogue, so a way in that has not been - built would read as one every backend already offers. - - Args: - agents: The backends there are, by name. - - Returns: - One capability per way something here is reached through, with the backends it reaches - -- or none at all against a way every one of them serves. The roads a machine declares - are asked under `where=` and the two humanize reaches from inside a process it started - are asked of the agent, which is the split :data:`_ANCHORS` argues. - """ - held: list[Capability] = [] - for name, said in _ANCHORS.items(): - road = name in _REACHED - reaching = frozenset(agents) if road else _tagged(name, agents) - if not reaching: - continue - held.append( - Capability( - name, - frozenset() if reaching == frozenset(agents) else reaching, - said, - WHERE if road else OF_AGENT, - ) - ) - return held - - -def _adds() -> list[Capability]: - """The settings one backend's config takes that the common one does not, one name apiece. - - Every backend is configured with the same handful of things -- a model, an effort, a rung, - an account -- and some of them take more than that: a CLI with a flag nobody else has is - driven by a subclass of :class:`hmz.coganchor.agents.config.AgentConfig` carrying a field - for it. That field is where humanize's own imposition on a CLI's defaults is written down - and where it is undone, so it is worth asking for before an agent is chosen: a run that - means to pin a turn's own clock, or to let a project be resolved the way the bare CLI - would, needs the backend whose config has somewhere to say so. - - Read off the config classes rather than declared beside them, for the reason everything - else here is read off the live interface: a field added to one of them is a name this - answers with on the next call, and a field renamed cannot leave a capability behind - promising what nothing serves. - - Returns: - One capability per such field, named `settings:`, against the backends whose own - config carries it. - """ - import dataclasses - - from hmz.coganchor.agents import DRIVEN - from hmz.coganchor.agents.config import AgentConfig - - common = {one.name for one in dataclasses.fields(AgentConfig)} - own = { - backend: {one.name for one in dataclasses.fields(config)} - common - for backend, (_, config) in DRIVEN.items() - } - return [ - Capability( - f"settings:{field}", - frozenset(backend for backend, fields in own.items() if field in fields), - f"a setting only some of these CLIs have -- replace(agent.config, " - f"{field}=...) on a backend among these, whose own config class is what " - f"carries it and whose default is what a turn of it already ran as", - ) - for field in sorted({field for fields in own.values() for field in fields}) - ] - - -def catalogue() -> tuple[Capability, ...]: - """Everything a flow may build on here, read off the installed interface at call time. - - At call time rather than written down, because this is what keeps a generated flow - honest across versions: the moments come off the live enum, the backend sets off the - driver classes' own declarations, and the asks off the same interfaces `surface` reads - -- so what the catalogue promises is what this installation serves, not what some - edition of it once did. - - Read here rather than in `hmz.coganchor`, deliberately, and the line is between the fact - and the ask. A capability name is both: something a backend, a driver or a machine - declares, and something a flow writes down where it declares a place. The fact belongs - wherever the thing it is about is written down -- a CLI's in `backends.py`, a driver's on - the driver class, a machine's in :mod:`hmz.coganchor.places` -- and every one of them is - read live from there rather than copied. The ask belongs here, because every reader of it - is here or above: `comes_to`, `serves`, `briefed`, the picker. A layer whose whole job is - driving a coding agent has no business knowing what a flow may write beside a place, and - what used to be wrong was not the address but the contents: the machine vocabulary was - written out in this file under names `hmz.coganchor.machines` was separately writing out - for itself, which is one word in two places. - - One word per capability, and where the capability is a setting that word is the derived - one: what comes to exactly "this backend's own config carries this field" is served as - `settings:` and under nothing else. A hand-written name beside it would be the - same fact in two places, going on being right in only one of them the day the field is - renamed, and a flow generated against either word asks under whichever it happened to - read. What earns a word of its own is a question the field's presence does not answer: - `narrate` is whether a session can be told to narrate at all, which two configs carry - `partial_messages` for and one session does; `tier:fast` is `service_tier`, a field of - the common config that every backend has somewhere to say and only three can serve. - - `pursue` and `goal` are one answer under two words and are meant to be. They are not the - same fact twice the way a hand-written `trust` beside `settings:trust` would be: one asks - whether `session.pursue(objective)` can be called and the other whether the place may be - declared `Annotated[Agent, Goal]`, which are two things a flow writes in two different - files, and neither is derived from a name that could be renamed out from under it -- both - are read from `cls.pursues` here, in one place, so a rename orphans neither. The rule the - settings block states is about *derived* names, and this is not one. - - Returns: - One capability apiece, each saying which half of - :class:`~hmz.coganchor.agents.config.Needs` asks for it. Asked of the agent: the - primitives every backend serves, then what only some do -- each moment outside - `EVERYWHERE`, the shape a turn can be held to, the tools a flow may offer, a turn that - can be steered while it runs, the goal feature, the fork, the CLI's own search, fleet - and resume, each kind of token a backend says what it spent on, the turn that can be - told to say what it is reaching for as it writes it, the faster tier some can be asked - to serve at, each rung some can be held to, each setting only some of their configs - carry, and the two roads humanize reaches a turn down from inside a process it started. - Asked of where it works: where an agent's turns may land, and the roads an anchor - reaches a machine by. - """ - import inspect - import sys as running - - from hmz.coganchor.agents import ( - DRIVEN, - EVERYWHERE, - KINDS, - PERMISSIONS, - Moment, - rung, - ) - - agents = {name: held[0] for name, held in DRIVEN.items()} - sessions: dict[str, type] = {} - for name, cls in agents.items(): - # The session class, read off what `new` says it answers with: the class itself - # is what carries `shapes` and `takes_tools`. By the name in the driver's own - # module rather than through `get_type_hints`, which would ask every annotation - # in the signature to resolve -- and a driver is free to keep `os` under - # TYPE_CHECKING. - told: object = None - with contextlib.suppress(Exception): - told = inspect.signature(cls.new).return_annotation - if isinstance(told, str): - told = vars(running.modules[cls.__module__]).get(told) - if isinstance(told, type): - sessions[name] = told - held: list[Capability] = [ - Capability( - "turns", - frozenset(), - "one turn in a session of its own -- agent(prompt, suppress=True) -- or many " - "at once with agent.batch(prompts); suppress=True makes a failed turn answer " - "'' (or None, for a shaped one) instead of raising", - ), - Capability( - "sessions", - frozenset(), - "one conversation held across turns -- session = agent.new(cwd=...) then " - "session(prompt) -- and dropping the session is how a flow forgets", - ), - Capability( - "schema", - frozenset(), - "a turn read back as an object -- session(prompt, suppress=True, " - "schema=Model) answers the model or None, and the answer is guarded before a " - "field is read off it", - ), - Capability( - "budgets", - frozenset(), - "what a run is held to, which is not the flow's to implement -- every run has " - "an allowance in hours, millions of output tokens and dollars, a turn taken " - "under a spent one raises Stopped, and a flow may declare its own default with " - "@flow(budget=Allowance(...)). What a flow may still read as it goes: " - "agent.spent().output climbs as the run spends, and session.rate() and " - "session.juice() say how fast", - ), - Capability( - "hooks", - frozenset(), - "a word in at the moments of a turn -- agent.hooks.on(Moment.STOP, hook) " - "hangs a callable, and a Stop hook that refuses sends the agent on with what " - "it said", - ), - Capability( - "subflows", - frozenset(), - "one flow runs another -- load('official/rlar')(agents, task) -- found by " - "the same name -f takes, and refused where it is asked for if nothing " - "answers to it", - ), - Capability( - "person", - frozenset(), - "the person at the prompt is a place like any other -- agents that include a " - "Person field -- and person(said) asks them; run where nobody is at a " - "prompt, they answer nothing, and a flow written to stop on nothing stops", - ), - Capability( - "board", - frozenset(), - "named lines the flow and the person both write on and neither waits at -- " - "person.board.put('todo', said) and person.board.get('todo')", - ), - Capability( - "state", - frozenset(), - "a flow marked @flow(resumable=True) is handed a dict as its last argument, " - "holding what it wrote there last time -- and clears it when the run is over", - ), - Capability( - "config", - frozenset(), - "what a flow can be set up with is a pydantic model third argument -- " - "config: Config | None = None -- whose model_config refuses extras and whose " - "every field carries Field(description=...)", - ), - Capability( - "skills", - frozenset(), - "a flow's own skills live in its skills/ directory, one directory per skill " - "with a SKILL.md in it, and a session says which it carries with " - "session.loads([...])", - ), - Capability( - "clone", - frozenset(), - "an agent set up differently is another agent -- " - "agent.clone(config=replace(agent.config, effort='high')) -- having opened " - "nothing and spent nothing", - ), - Capability( - "moments", - frozenset(), - "the moments every backend reaches, for a hook to hang on: " - + ", ".join(f"Moment.{one.name}" for one in Moment if one in EVERYWHERE), - ), - ] - for moment in Moment: - if moment in EVERYWHERE: - continue - held.append( - Capability( - f"moment:{moment.value}", - frozenset( - name for name, cls in agents.items() if moment in cls.moments - ), - f"Moment.{moment.name} is reached only where the backend says so -- " - f"declare it on the place, Annotated[Agent, Moment.{moment.name}], and " - "the run is refused an agent that cannot run it before its first turn", - ) - ) - held.append( - Capability( - "shape", - frozenset( - name for name, one in sessions.items() if getattr(one, "shapes", False) - ), - "a turn held to the shape rather than asked to keep to it -- every backend " - "takes schema=, and these are the ones the answer is certain on", - ) - ) - held.append( - Capability( - "tools", - frozenset( - name - for name, one in sessions.items() - if getattr(one, "takes_tools", False) - ), - "a flow's own callbacks put in front of the agent -- " - "session.offers([Tool(...)]) -- which a backend not among these refuses; " - "type(session).takes_tools says so beforehand", - ) - ) - held.append( - Capability( - "steer", - frozenset( - name for name, one in sessions.items() if getattr(one, "steers", False) - ), - "a word put into the turn already running -- session.interject(said) -- which " - "the agent takes into that turn rather than answering as the next one; " - "type(session).steers says so beforehand, and a backend not among these " - "refuses rather than queueing it behind", - ) - ) - pursuing = frozenset(name for name, cls in agents.items() if cls.pursues) - held.append( - Capability( - "pursue", - pursuing, - "the backend's own goal feature -- session.pursue(objective) keeps the " - "agent going until it decides for itself the objective is met", - ) - ) - held.append( - Capability( - "goal", - pursuing, - "a place run under that feature declares it -- Annotated[Agent, Goal] -- " - "and is refused an agent whose backend has none before the first turn", - ) - ) - held.append( - Capability( - "fork", - _tagged("fork", agents), - "one conversation carried into a second going its own way -- " - "child = session.fork() -- which costs nothing until the child is used, and " - "which session.forks says of a backend beforehand", - ) - ) - for name, said in _PROFILED.items(): - # Both normalisations `_anchors` makes, and for the same two reasons. A fact no CLI - # here has is left out rather than listed against nobody, an empty set being how this - # catalogue says "all of them"; and one every CLI here has is listed against nobody - # rather than against twelve names, so that it reads as universal on the briefing and - # so that a CLI somebody added by hand -- which is in no `DRIVEN` table -- is not read - # as lacking what every backend has. - serving = _tagged(name, agents) - if not serving: - continue - held.append( - Capability( - name, - frozenset() if serving == frozenset(agents) else serving, - said, - ) - ) - for permission in PERMISSIONS: - # Every driven backend reads as none of them, the way an anchor every CLI is reached - # through does: an empty set is what says "all of them", and it is also the only - # answer that reaches a CLI somebody added by hand -- one driven over the protocol is - # in no `DRIVEN` table, so a rung listed against every name there is a rung that - # backend would read as not served. `bypass` is exactly that rung. - taking = frozenset( - name for name, cls in agents.items() if permission in cls.rungs - ) - held.append( - Capability( - rung(permission), - frozenset() if taking == frozenset(agents) else taking, - f"the agent held to {permission!r} -- AgentDefaults(permission=" - f"{permission!r}) beside the place -- which a backend not among these refuses " - "where the agent is made rather than running it a rung looser; how coarsely a " - "backend that takes the rung maps it onto its own settings is another " - "question and is written down in docs/reference/agents.md", - ) - ) - held.extend( - Capability( - f"counts:{kind}", - frozenset(name for name, cls in agents.items() if kind in cls.counts), - f"the backend says what a turn spent on {kind} tokens -- " - f"agent.spent()['{kind}'] and session.rate()['{kind}'] answer for it, and a " - "backend not among these reports the kind not at all rather than as nothing, so " - "a run that mixes one in reads its figure for this kind as a floor", - ) - for kind in KINDS - ) - held.append( - Capability( - "tier:fast", - frozenset( - name - for name, cls in agents.items() - if "fast" in getattr(cls, "service_tiers", ()) - ), - "the provider asked to serve this agent's turns faster at the same model and " - "the same effort -- AgentConfig(service_tier='fast') -- which each of these " - "sends in its own backend's word for it; a backend not among them refuses the " - "tier before the first turn rather than quietly running at another one", - ) - ) - held.append( - Capability( - "narrate", - frozenset( - name - for name, one in sessions.items() - if getattr(one, "narrates", False) - ), - "a turn that can be told to say what it is reaching for while the arguments " - "are still being written -- a long write announced as it happens rather than " - "once the file is in the call -- so something watching a turn can tell one that " - "is working from one that has wedged; type(session).narrates says so " - "beforehand, and ClaudeCodeAgentConfig(partial_messages=False) turns it off for " - "one agent", - ) - ) - held.extend(_adds()) - held.extend(_places()) - held.extend(_anchors(agents)) - return tuple(held) - - -def misplaced(names: Iterable[str], asked: str) -> frozenset[str]: - """Which of these names are asked for in the other half of `Needs` than the one they are in. - - The one slip in this vocabulary that nothing used to catch. `Needs("isolated")` and - `Needs(where=("steer",))` are both legal Python, both spell a name that exists, and both - were answered by the wrong side: the first passed whatever filled the place, because a - machine capability carries no backends and no backends means every backend; the second was - refused by every machine there is, because no machine's settings have ever carried a word - about the agent. One always yes and one always no, and neither of them a measurement. - - A name the catalogue does not have at all is not misplaced and is not named here: it is - something nothing serves, and the refusal that says so is the ordinary one. - - Args: - names: What the place asked for, in one half of :class:`~hmz.coganchor.agents.config.Needs`. - asked: Which half that is -- :data:`OF_AGENT` or :data:`WHERE`. - - Returns: - Those of them the catalogue asks for in the other half. - """ - other = {one.name for one in catalogue() if one.asked != asked} - return frozenset(other & set(names)) - - -def briefed() -> str: - """The catalogue rendered as the one page a compiler steers by. - - Three sections rather than two, because a flow writes the vocabulary in two places and - reading it as one is how `Needs("isolated")` came to be written: what a machine has to - come to goes in `where=` and what the backend has to serve goes beside it, and a page that - ran the two together was teaching the mistake. - - Returns: - What every backend serves, then what only some do -- each with the backends that - do, so that a flow built on one can say where it runs -- and then what a place may be - asked to come to. - """ - held = catalogue() - agent = [one for one in held if one.asked == OF_AGENT] - lines = [ - "What a flow may build on here, read off this installed humanize.", - "", - "Every backend -- Needs(...) beside the place:", - ] - lines.extend(f"- {one.name}: {one.said}" for one in agent if not one.backends) - lines += [ - "", - ( - "Only some backends -- a flow built on one of these says so where it " - "declares the place, and is refused an unfit agent before the first turn:" - ), - ] - lines.extend( - f"- {one.name} ({', '.join(sorted(one.backends))}): {one.said}" - for one in agent - if one.backends - ) - lines += [ - "", - ( - "Where an agent's turns land -- Needs(where=(...)) beside the place, answered " - "by that machine's own settings rather than by any backend, and refused before " - "an image has been pulled:" - ), - ] - lines.extend(f"- {one.name}: {one.said}" for one in held if one.asked == WHERE) - return "\n".join(lines) diff --git a/src/hmz/runtime/flowing/driving.py b/src/hmz/runtime/flowing/driving.py deleted file mode 100644 index 06c42d1d..00000000 --- a/src/hmz/runtime/flowing/driving.py +++ /dev/null @@ -1,2613 +0,0 @@ -"""What a flow declares, what it brings, and how one is run by another. - -A flow is a Python file, and the only thing it has to say about itself is what it drives: how -many agents, what each is for, what each has to be able to do and where each may work. That is -read here, off the annotation on the entry point, so that whatever is starting a flow -- a -command line, the flow picker, another flow -- can put the right questions before anything -runs. A flow given the wrong number of agents, or one that cannot run a moment the flow hangs -a hook on, is refused where it was written down rather than hours into a loop. - -And a loop worth having is one another loop can reach for, which is :func:`load`: a flow -found by the same name `-f` takes, handed the agents the calling flow was given, carrying its -own skills and its own kept state, and written down -- in a record of its own, inside the -record of the flow that called it -- as running under whatever called it. - -A run of flows calling flows is a tree rather than a list, and it is tracked as one. Each -flow says which flow called it, and which that was is read off the task the call was made -from rather than off the process: a flow written as a coroutine may gather two calls at once, -those two run at the same moment on one thread, and neither of them is under the other. So a -call made from inside either lands under the one it was made from, a session opened inside it -is written into that one's record, and what is running, read from inside a flow, is the -branch that flow is on and not everything the run happens to be doing. - -Nothing here reads a command line and nothing here opens an epic: :mod:`hmz.runtime.runner` does -both, and asks this what the flow it was named says about itself. A call asks the epic already open -for a record to be written into, which is not a second epic: it is part of the one run. -""" - -from __future__ import annotations - -import contextlib -import contextvars -import inspect -import os -import threading -import time -from pathlib import Path -from typing import ( - TYPE_CHECKING, - Annotated, - Any, - Final, - NamedTuple, - cast, - get_args, - get_origin, - get_type_hints, -) - -from hmz.runtime import telemetry - -if TYPE_CHECKING: - from collections.abc import ( - Awaitable, - Callable, - Generator, - Iterable, - Mapping, - Sequence, - ) - - from pydantic import BaseModel - - from hmz._legacy_flows import Agent, Driven - from hmz._legacy_flows import Flow as Marked - from hmz.coganchor.agents import ( - AgentBase, - AgentConfig, - AgentDefaults, - Isolated, - Moment, - Needs, - Remote, - ) - from hmz.coganchor.agents.allowance import Allowance - from hmz.coganchor.agents.base import Journal - from hmz.coganchor.agents.skills import Loaded - from hmz.coganchor.machines import MachineBase, MachineConfig, Mapped - from hmz.runtime.epic import Epic, Sub - - from .checking import Capability - -__all__ = [ - "Entry", - "NotAFlow", - "Place", - "Running", - "carries", - "comes_to", - "configures", - "contained", - "container", - "declared", - "declares", - "drives", - "entered", - "lands", - "lands_in", - "left", - "load", - "readies", - "resumes", - "running", - "runs_at", - "serves", - "set_up", - "wanted", -] - - -#: How many arguments a flow's entry point takes when it says it can be set up with -#: something: the agents, the task, and the model that says what there is to set. -_WITH_A_CONFIG = 3 - -#: A flow's entry point: called with the agents and the task, and done when it returns -- -#: or, for one written as `async def run`, when what it returns has been awaited. Which of -#: the two a flow is, is the flow's own business: `Runner.run` waits for it either way. -type Entry = Callable[..., Awaitable[None] | None] - - -class NotAFlow(ValueError): # noqa: N818 -- the name the SPEC gives it - """What a command line named, when it was not a flow for the agents it was given. - - Its own kind of error, so that a flow failing as it is imported -- one that reads a prompt - file beside it and does not find it -- is left to fail as it would anywhere, rather than - being reported as a command line to correct. - """ - - -class Running(NamedTuple): - """One flow that is running now, on the branch of the run it is running on. - - Attributes: - flow: What it was asked for as -- the name a command line gave, or the one a flow asked - another for, which is the name worth showing either way. - since: When it started, on the monotonic clock. - depth: How far down the calls it is: 0 for the flow somebody started, one more for each - flow that had to be called to reach it. - under: The flow that called it, or None for the one somebody started. What makes a run - a tree rather than a list: two flows gathered at once run at the same moment and - neither of them is under the other, so only the flow each was called from can say - where it belongs. - """ - - flow: str - since: float - depth: int = 0 - under: Running | None = None - - -class _Held(NamedTuple): - """One flow of a run, as whatever is watching the run needs it. - - Attributes: - one: What it is, and what called it. - thread: The thread `entered` was called on, which is what says it is still there. - agents: What it is being driven with, for a report of a failure in it. - """ - - one: Running - thread: threading.Thread - agents: Sequence[Agent] - - -#: Every flow running now, in the order they started: the one somebody ran, then whatever any -#: of them called, each beside the thread it was entered on and the agents it drives. Kept -#: here rather than asked of the flows, which is the one thing a flow cannot be asked -- it is -#: a Python file and may branch any way it likes -- and read by the interface to say what is -#: running under what. -#: -#: Flat, with the shape carried by the records themselves: each one says what called it. It -#: cannot be a stack, because a flow written as a coroutine may have two calls going at once -#: on one thread and neither of those is under the other. Keyed by the identity of the record -#: `entered` made, so that what goes with a flow goes when that flow does: a crash in the -#: interface an hour after a flow ended must not be filed as a crash in that flow, and a flow -#: that called another must not have the called one's agents put under its name. -#: -#: Under a lock, since a flow runs on whichever thread took it and the interface reads while -#: they run. -_RUNNING: dict[int, _Held] = {} -_TELLING = threading.Lock() - -#: The flow this task is inside, which is the branch a call made from here goes on. A context -#: variable rather than one slot for the thread or for the process: `asyncio` copies the -#: context into every task it starts, so two flows gathered at once each get one of their own -#: and a third called from either lands under the one that called it rather than under -#: whichever of them happened to start last. A thread that is not running a flow -- an -#: interface drawing a status line, a turn taken on a thread of its own -- has none, and -#: reads the whole of the above instead. -_ON: contextvars.ContextVar[Running | None] = contextvars.ContextVar( - "hmz_flow_running", default=None -) - - -def running() -> tuple[Running, ...]: - """Every flow running now: the branch this is asked from, or all of them from outside. - - Asked from inside a flow -- by the flow's own code, by a hook it hung, by a callback an - agent reached for -- it answers with the branch that flow is on: the one somebody - started, then each flow that was called to get here, innermost last. That is what a flow - can truthfully be told, and the only thing a flow gathering two calls at once can be: - the sibling running beside it is not running under it and is none of its business. - - Asked from outside one -- the interface drawing a status line, a report being written -- - it answers with every flow of the run, oldest first, each saying how deep it is and what - called it. A run is a tree, and from outside there is no branch to be on. - - A flow says it has ended as it ends, however it ends -- but only a flow that got the - chance to. One whose thread has gone was abandoned where it stood rather than finished: - an interface taken down under it, a test that let go of it. So what is running is checked - against the threads running it, and a flow with no thread left is not one of them. - - Returns: - One apiece. Empty where nothing is running. - """ - with _TELLING: - for key in [ - key for key, held in _RUNNING.items() if not held.thread.is_alive() - ]: - del _RUNNING[key] - here = _ON.get() - if here is None: - return tuple(held.one for held in _RUNNING.values()) - branch = list[Running]() - while here is not None: - branch.append(here) - here = here.under - # Only the ones still running: a branch is read off records that hold each other, - # and one whose thread has gone was pruned above rather than left to be reported. - # A branch with nothing left on it is nothing running here, and never everything - # running anywhere -- what a flow must not be handed is its siblings. - return tuple(one for one in reversed(branch) if id(one) in _RUNNING) - - -def entered(flow: str, agents: Sequence[Agent] = ()) -> Running: - """Writes down that a flow has started here, under whatever called it here. - - Under whatever called it *in this task*, which is what makes a run a tree: a called flow - runs on the branch of the flow that called it, and two called at once run on two branches - of it. What this task is inside is written down as well as answered with, so that a flow - called from inside this one lands under it. - - Args: - flow: What it was asked for as. - agents: What it is being driven with, for a report of a failure in it. - - Returns: - The record, to be handed back when it ends. - """ - under = _ON.get() - one = Running( - flow, time.monotonic(), 0 if under is None else under.depth + 1, under - ) - with _TELLING: - _RUNNING[id(one)] = _Held(one, threading.current_thread(), agents) - _ON.set(one) - return one - - -def left(one: Running) -> None: - """Writes down that a flow has ended, however it ended. - - What this task is inside goes back to whatever called that flow, rather than to a token - taken when it started: a flow written as a coroutine is entered where the call was - written and left inside the task that ran it, and those are two contexts -- while the - branch it was on is the same one read from either. - - Args: - one: What :func:`entered` answered with. - """ - with _TELLING: - _RUNNING.pop(id(one), None) - if _ON.get() is one: - _ON.set(one.under) - - -#: The container this run is in, or None for a run on this machine. One per process rather -#: than one per flow: a flow that called another is one run, working in one place, and two -#: containers under one run would be two workspaces the second flow could not see the first's -#: work in. Set by `contained` while the run is being got ready and taken down when it ends. -_INSIDE: list[tuple[MachineBase, MachineConfig, Mapped]] = [] - -#: Held over every look at the list above and every change to it, so that one run at a time -#: is settled rather than raced: two started at once would otherwise both have found nothing -#: there and gone ahead. Held for the look and not for the bringing up, which is a pull of -#: minutes -- the image below is what says the place is taken while that is going on. -_ENTERING = threading.Lock() - -#: The image of a container that is on its way up, for as long as that takes. A run is in one -#: from the moment another would have to wait for it rather than from the moment it answers, -#: so that the second of two is refused at once instead of at the end of the first's pull. -_COMING: list[str] = [] - - -def container() -> Mapped | None: - """The container this run is working in, as the flow's own code reaches it. - - A run may be put in a container of its own, which puts every one of its agents there: the - project directory is mounted at the path it already has, so a file the flow opens is the - same file a turn opened, and only the tools differ. What a mounted directory does not - answer for is a command -- one the flow runs is run by this machine's shell against this - machine's tools -- so that is what this is for:: - - if (held := flows.container()) is not None: - held.run(["pytest", "-q"]) - - Returns: - The workspace on the machine the run lands on, or None for a run on this machine -- - where a flow does what it always did, since the tools it would reach for are this - machine's either way. - """ - return _INSIDE[0][2] if _INSIDE else None - - -@contextlib.contextmanager -def contained(image: str, workspace: str = "") -> Generator[MachineConfig | None]: - """Puts a whole run in one container, for as long as the block lasts. - - Which is the convenience it is: an agent may already be pointed at a machine one at a - time, and a run that wanted all of them in a container was a flow declaring `Isolated` - beside every place it drives. This is that said once, from outside the flow -- the - container is started here, every agent is pointed at it, and the flow's own reads, - writes and commands reach it through :func:`container`. - - One container for the run rather than one per agent, which is the whole point: the agents - are working on one thing, so what one of them writes is what the next one reads. And one - at a time in a process for the same reason, the container being the process's own rather - than any one run's: two started at once with an image between them would be two runs - reaching for one container, so the second is refused rather than handed the first's. - - Args: - image: The image to run, which needs a `python3` for coganchor's target half and - whatever else the run expects its agents to reach for. "" is a run on this machine, - which starts nothing at all. - workspace: The project directory to give it, defaulting to this one. It is the - directory itself that goes there rather than a copy, at the path it already has, so - the work outlives the container. - - Yields: - The machine every agent of the run is to be pointed at, or None for a run on this - machine. - - Raises: - FileNotFoundError: If there is no workspace to give it, or no `docker` to give it to. - RuntimeError: If a run of this process is in a container already, or if the container - cannot be started. - """ - if not image: - yield None - return - from hmz.coganchor.machines import AnchoredConfig, DockerConfig, Mapped - - with _ENTERING: - if _INSIDE or _COMING: - raise RuntimeError( - "a run of this process is in a container already, and the container is " - "the process's rather than any one run's -- so a second run started beside " - "it would be reaching for the first's" - ) - _COMING.append(image) - try: - machine = DockerConfig(image=image, workspace=workspace or None).create() - # Started once and named by the anchor that reaches it, so that every agent of the - # run is pointed at the container that is already up rather than starting one apiece. - anchor = machine.start() - held = Mapped(anchor) - where_ = AnchoredConfig(anchor=anchor) - except BaseException: - with _ENTERING: - _COMING.clear() - raise - with _ENTERING: - _INSIDE.append((machine, where_, held)) - _COMING.clear() - try: - yield where_ - finally: - with _ENTERING: - _INSIDE[:] = [one for one in _INSIDE if one[0] is not machine] - held.close() - machine.stop() - - -def lands_in(agents: Sequence[Agent], where_: MachineConfig) -> None: - """Puts every agent of a run in the container the run is working in. - - Over whatever each was configured with, because that is what asking for a run to be in a - container means: it is said once, from outside, about all of them. An agent the flow - itself put in a container of its own is left where the flow put it -- where an agent works - is the flow's to say, and this is a convenience rather than a way round that -- and so is - the person at the prompt, who takes no turn anywhere. - - Args: - agents: The agents of the run, the person among them. - where_: The machine they are all to land on. - - Raises: - RuntimeError: If one of them has already opened a conversation, which is a conversation - that cannot be moved. - """ - from hmz.coganchor.agents import HumanAgent - from hmz.coganchor.machines import DockerConfig - - for one in agents: - if isinstance(one, HumanAgent) or isinstance(one.config.machine, DockerConfig): - continue - cast("Driven", one).runs_on(where_) - - -class Place(NamedTuple): - """One of the agents a flow drives, as the flow's own annotation declared it. - - Attributes: - name: What the flow calls it, or "" for a flow that said how many it drives and no more. - person: Whether it is the person at the prompt, who is handed over rather than chosen. - moments: The moments the agent filling it has to run, which the flow said by writing - `Annotated[Agent, Moment.PERMISSION_REQUEST]` where it declared the place. Empty - where it asked for nothing in particular, which is most places. - goal: Whether the flow runs this one under the backend's own goal feature, which it - said by writing `Annotated[Agent, Goal]` where it declared the place. Only four - backends have one, so a flow built on it is not a flow any agent can drive. - where: Where the agent filling it may work, which the flow said the same way -- `Remote` - for one that may be pointed at another machine, an `Isolated` for one that works in a - container the flow itself names the image of. None for a place the flow said nothing - about, which runs here and may not be sent anywhere: a flow is written for a shape of - work, and where its agents work is the flow's to say rather than a setting somebody - reaches for. - permission: What the agent filling it may do without being asked, which the flow said - with `AgentDefaults(permission=...)` beside the place. - :data:`~hmz.coganchor.agents.UNSAID` for a place that said nothing -- not a rung but - the absence of one, humanize saying nothing to the CLI about what its agent may do, - and looser than the loosest rung there is. So it settles nothing, and a flow with no - opinion about permissions leaves the agent filling the place with whatever it came - with, which is itself most often the same silence. What an agent already carries is - never loosened to reach any of these. - goals: Whether the backend's own goal feature is available to it, said the same way. - A place run under a `Goal` has them, whatever else it wrote. - web_search: Whether it may search the web, said the same way, and None for a place that - said nothing either way. - insist: Whether a backend that cannot carry one of the three refuses the run, said the - same way and True for a place that said nothing -- which is every place written - before there was a word for it, and what the doctrine asks for. False settles what - the backend can be told, drops the one setting it cannot, and says which place, which - setting and what the agent does instead. It is never the features the flow calls: - those are refused whatever this says, an agent without `steer` being one whose loop - breaks at the first `interject` rather than before its first turn. - needs: What filling this place takes, which the flow said by writing - `Annotated[Agent, Needs("steer", where=("isolated",))]` where it declared it -- what - the backend has to serve, and what the machine its turns land on has to come to. - None for a place the flow said nothing about, which is most of them: what every - backend serves and every machine holds is nothing a flow has to ask for. - """ - - name: str - person: bool - moments: frozenset[Moment] - where: type[Remote] | Remote | Isolated | None = None - goal: bool = False - # `UNSAID` under its own name lives in `hmz.coganchor.agents.config`, and is spelled out - # here rather than imported: a flow is read long before any of its agents is started, and - # naming it would make reading one pay for the half of coganchor that runs a session. The - # word is the empty string in both places for exactly that reason -- nothing said is the - # same nothing wherever it is read, and is falsy where either of them asks. - permission: str = "" - goals: bool = True - web_search: bool | None = None - insist: bool = True - needs: Needs | None = None - - -def drives(flow: str | os.PathLike[str]) -> tuple[str, ...]: - """What a flow calls each of the coding agents it drives, in the order it takes them. - - Read without being given any, so that a caller can ask before it has them -- which is - what choosing the agents for a flow means. - - Args: - flow: The Python file the flow is written in. It is run to be read. - - Returns: - One name per agent its entry point declares that somebody has to choose, which is how - many it has to be given. A flow that declares a plain tuple has not named them, and each - is "" -- the count is all it said. A place it declared as a - :class:`~hmz._legacy_flows.Person` is not - among them: nobody chooses what the person at the prompt runs, so nobody is asked. - - Raises: - NotAFlow: If the file is not there, or is not a flow. - """ - return tuple(place.name for place in wanted(flow)) - - -def configures(flow: str | os.PathLike[str]) -> type[BaseModel] | None: - """What a flow can be set up with before it is run, if it takes anything at all. - - A flow says so by taking a third argument annotated with a pydantic model or None: the - model is the whole of what may be asked, since the fields, their types, what each one is - for and the combinations the flow refuses are already written down in it. So whatever is - starting a flow can put the questions to somebody without knowing what any of them mean. - - Args: - flow: The Python file the flow is written in. It is run to be read. - - Returns: - The model to ask with, or None for a flow that takes the agents and the task and - nothing else -- which is most of them, and is what every flow was before this. - - Raises: - NotAFlow: If the file is not there, or is not a flow. - """ - return declares(flow)[3] - - -def resumes(flow: str | os.PathLike[str]) -> bool: - """Whether a flow says it can be picked up where the last run of it left off. - - Which is a thing about the flow rather than about any run of it: a run wrote down what - the flow said when it ran, and the flow may have been rewritten since -- so whatever is - offering to pick a run up asks the flow as it is now. - - Args: - flow: The Python file the flow is written in. It is run to be read. - - Returns: - True for a flow marked `@flow(resumable=True)`, which is one handed the state of the - last run of it. - - Raises: - NotAFlow: If the file is not there, or is not a flow. - """ - return declares(flow)[4].resumable - - -def declared(flow: str | os.PathLike[str]) -> Allowance | None: - """What a flow says a run of it may spend by default, if it says anything. - - A default and never an implementation: a flow does not hold itself to this, and whoever - starts a run of it overrides it. Asked by whatever is about to run one, so that a flow - rewritten since the last run of it is read as it is now. - - Args: - flow: The Python file the flow is written in. It is run to be read. - - Returns: - What `@flow(budget=...)` said -- `Allowance()` for a flow saying it is meant to run - under nothing at all, and None for one with no opinion, which is the difference - between a deliberate unbounded run and an accidental one. - - Raises: - NotAFlow: If the file is not there, or is not a flow. - """ - return declares(flow)[4].budget - - -def _marked(run: Entry) -> Marked: - """What a flow said about itself where it was marked. - - Args: - run: Its entry point, which is what carries the mark. - - Returns: - The mark. Never None: only a marked function is a flow, so anything that got this far - has one -- and a flow whose mark cannot be read is read as one that said nothing. - """ - from hmz._legacy_flows import Flow as Said - - held = getattr(run, "__humanize_flow__", None) - return held if isinstance(held, Said) else Said() - - -def wanted(flow: str | os.PathLike[str]) -> tuple[Place, ...]: - """Every agent a flow needs chosen for it, and what each of them has to be able to do. - - What :func:`drives` says, and what the flow asked of each place besides a name: a flow - that hangs a hook on a moment only some backends run says so in the annotation, and - whoever is choosing the agents can then offer only the ones that would work. - - Args: - flow: The Python file the flow is written in. It is run to be read. - - Returns: - One place per agent somebody has to choose, in the order the flow takes them. - - Raises: - NotAFlow: If the file is not there, or is not a flow. - """ - return tuple(place for place in declares(flow)[1] if not place.person) - - -def declares( - flow: str | os.PathLike[str], -) -> tuple[ - Entry, - tuple[Place, ...], - Callable[..., tuple[Agent, ...]], - type[BaseModel] | None, - Marked, -]: - """Loads a flow and reads what it says about the agents it drives. - - Args: - flow: The flow: one that came with humanize, by name, or a file of your own. - - Returns: - Its entry point, one place per agent it drives, what to hand those agents over as -- - the named tuple the flow declared, or a plain one where it declared that -- the model - it can be set up with, or None where it takes no setting up, and what the flow said - about itself where it was marked. - - Raises: - NotAFlow: If the file is not there, is not a flow -- nothing in it marked `@flow()`, or - one whose `agents` cannot be read or says nothing about how many it takes. - """ - from .finding import find, inside, loaded - - named = str(flow) - # Which of the file's flows was asked for, before the name is resolved to a file: a file - # may hold several, and `humanize1:gen-plan` is one of them. - wanted = inside(named) - # Resolved here rather than by whoever is starting one, so that a name works wherever a - # flow is named -- a command line, an interface, a `Runner` written by hand. - flow = find(named) - # The same test `find` applies, and for the same reason: a place that cannot be read - # holds no flow, which `Path.is_file` would raise about rather than answer. - if not os.path.isfile(flow): # noqa: PTH113 - raise NotAFlow(f"{flow}: {_unfetched(str(flow))}") - read = loaded(flow) - run = _entry(read, wanted) - if run is None: - # A file that holds several flows names each of them after itself, so whoever asked - # for the file alone -- or for one of them under a name it does not have -- is a - # colon away from what they meant, and saying which ones is what ends it. - holds = [f"{_called(flow)}:{one}" for one in _holds(read)] - missing = f"{flow}: nothing in it is a flow called {wanted!r}" - if wanted and holds: - raise NotAFlow(f"{missing}; it holds {', '.join(holds)}") - if wanted: - raise NotAFlow(missing) - if holds: - raise NotAFlow( - f"{flow}: nothing in it is marked @flow(), and it holds " - f"{', '.join(holds)} -- name the one to run" - ) - raise NotAFlow( - f"{flow}: nothing in it is marked @flow() -- a flow is a function marked with " - "it, which is how a file says which of the functions in it is one" - ) - try: - # A function, so that what is read below is what the entry point will be called - # with: a class or a partial answers with annotations that are somebody else's. - # Extras and all: what a flow wrote beside the type is what it asks of the agent. - hinted = ( - get_type_hints(run, include_extras=True) if inspect.isfunction(run) else {} - ) - declared = hinted.get("agents") - except NameError as unresolved: - # A flow whose agents are imported under TYPE_CHECKING states how many it drives - # where nothing can read it back, which is the one thing a flow is asked to say. - raise NotAFlow( - f"{flow}: the flow's agents cannot be read here ({unresolved}) -- import what " - "the annotation names at runtime, so the count it states can be checked" - ) from unresolved - # A named tuple is a tuple that also says what each of its places is for, and `_fields` - # is where it says it. `_make` builds one from a sequence, exactly as `tuple` does, so - # the flow is handed the type it asked for either way. - if ( - run is not None - and declared is not None - and (fields := getattr(declared, "_fields", None)) - ): - kinds = _kinds(declared, run) - return ( - _compiled(named, read, run), - tuple(_place(at, kinds.get(at)) for at in fields), - declared._make, - _setting(run, hinted), - _marked(run), - ) - # `tuple[Agent, ...]` is any number of them, which is no answer to the question. - declares = get_args(declared) - if run is None or get_origin(declared) is not tuple or Ellipsis in declares: - raise NotAFlow( - f"{flow}: a flow is a function marked @flow() taking (agents, task), whose " - "agents are annotated with a tuple of a fixed length -- how many agents the " - "flow drives -- or with a NamedTuple of them, which also says what each is for" - ) - return ( - _compiled(named, read, run), - tuple(_place("", kind) for kind in declares), - tuple, - _setting(run, hinted), - _marked(run), - ) - - -def _compiled(named: str, read: dict[str, Any], run: Entry) -> Entry: - """One flow's entry point, or -- for an atlas -- something that runs its prophecy. - - An atlas is a flow whose body is a declaration: what it says is compiled before anything - runs, and what runs is the prophecy that compiling made. So the entry point itself is - never called, and what everything else holds is the walk over the prophecy instead -- - swapped here, where a flow is loaded, so that every way of running one gets both the - compiling and the walking without knowing there are two kinds of flow. - - Args: - named: The flow, as it was asked for. - read: What running its file left behind. - run: Its entry point. - - Returns: - The entry point for an ordinary flow, and the walk for an atlas. - """ - from hmz._legacy_flows.atlas import ATLAS - - if getattr(run, ATLAS, None) is None: - return run - return _Walked(named, read, run) - - -class _Walked: - """An atlas's entry point, compiled when the run reaches it and not before. - - :func:`declares` is asked by everything that wants to know what a flow says as well as by - the two places that run one: how many agents it drives, what it can be set up with, - whether it can be picked up. Every one of those is answered off the entry point's own - annotation, and compiling the atlas to answer them would mean reading every file the flow - holds -- and, for one that does not compile, refusing a question the flow can answer. A - flow picker asking whether an atlas can be picked up would then be told no. - - So the compiling waits for the call, which is still before the first node runs: an atlas - is a flow checked before anything happens rather than one checked before anything is - read. Once compiled it is held, since this is one run of one flow. - """ - - def __init__(self, named: str, read: dict[str, Any], entry: Entry) -> None: - """Holds what it takes to compile one atlas, for the moment something runs it. - - Args: - named: The flow, as it was asked for. - read: What running its file left behind. - entry: The atlas's own entry point, which is never called. - """ - self._named = named - self._read = read - self._entry = entry - self._walk: Entry | None = None - # What the entry point was marked with, so that whatever reads a flow off what - # `declares` answered reads what it would have read off the entry point itself. - self.__dict__.update(entry.__dict__) - - def __call__(self, *said: Any) -> Awaitable[None] | None: - """Runs the atlas, compiling it first if this is the first call. - - Args: - said: What any flow is called with -- the agents, the task, the config for one - that takes one, and the dict a resumable flow is handed. - - Returns: - Whatever the prophecy answers with. - - Raises: - NotAFlow: If the atlas does not compile, saying each reason on a line of its own. - """ - return self.ready()(*said) - - def ready(self) -> Entry: - """Compiles the atlas, if this is the first thing to ask for it. - - Returns: - The walk over the prophecy. - - Raises: - NotAFlow: If the atlas does not compile, saying each reason on a line of its own. - """ - if self._walk is None: - from .stepping import walking - - self._walk = walking(self._named, self._read, self._entry) - return self._walk - - -def readies(run: Entry) -> Entry: - """Compiles whatever a flow has to have compiled before a run of it starts. - - An atlas is compiled when something reaches for the run rather than when a flow is read, - so that asking what a flow drives, what it can be set up with or whether it can be picked - up neither pays for a reading of every file it holds nor is refused by one. The two - places that are about to run one ask here instead: a body that does not compile is then - refused where the run is being set up, rather than from inside a run that has already - pulled an image and opened an epic. - - Args: - run: What :func:`declares` answered with. - - Returns: - The same thing, ready to be called. - - Raises: - NotAFlow: If it is an atlas that does not compile. - """ - if isinstance(run, _Walked): - run.ready() - return run - - -def _settles(agent: Agent) -> Driven: - """One agent as whoever hands it to a flow holds it, rather than as a flow does. - - A flow sees an agent through :class:`~hmz._legacy_flows.agent.Agent`, which is what a flow may - ask of one and says nothing about setting it up: an agent is what somebody already chose, - and a flow that could change it would be a flow rewriting that choice. This module is one - of the three places entitled to -- it settles where an isolated agent works, and what the - flow it is about to run works by -- so it says so here rather than reaching through a - class it must not name. - - Args: - agent: The agent, as the flow holds it. - - Returns: - The same agent, as whoever hands it over holds it. - """ - return cast("Driven", agent) - - -def carries(flow: str | os.PathLike[str], agents: Sequence[Agent]) -> None: - """Gives every agent of a flow the skills that flow works by. - - A flow is a directory, and what it keeps in `skills/` -- plus whatever it named where it - was declared -- is mounted onto every session these agents open. Told to the agents rather - than configured on them: the skills are the flow's, and the same agent under another flow - carries that flow's instead. - - Worked out here rather than at each session, because a repository named by a flow is - fetched to get it: a run that cannot reach one says so before the first turn rather than - an hour into a loop. Worked out afresh each time a flow is run or called, so that a flow - which has rewritten its own skills is driven by what it has now. - - Args: - flow: The flow, as it was named. - agents: The agents it is being run with. - """ - if (loaded := _brought(flow)) is None: - return - for agent in agents: - _settles(agent).loads(loaded) - - -def _brought(flow: str | os.PathLike[str]) -> tuple[Loaded, ...] | None: - """The skills one flow works by, read off the flow as it is on disk now. - - Worked out rather than mounted, so that what a called flow is to carry can be settled - before the call takes the agents: a call that is refused between the two must not leave - the flow that made it driving agents carrying somebody else's skills. - - Args: - flow: The flow, as it was named. - - Returns: - One apiece, or None for a flow that says nothing about skills at all -- which is a - flow that leaves the agents carrying whatever they carry rather than one that empties - them. - - Raises: - NotAFlow: If a repository the flow names cannot be reached. - """ - from .finding import at as directory - from .skills import brought - - where = directory(str(flow)) - declared: tuple[str, ...] = () - with contextlib.suppress(Exception): - # What the flow said where it was declared, which is read off the flow that was asked - # for. A flow that will not load is left to the loading to report. - declared = _brings(flow) - # A flow that is one file has no directory of its own and so brings no skills of its own - # -- but it may still name skills that live somewhere else, and those are as much what it - # works by as a directory flow's are. - if not where and not declared: - return None - try: - return tuple(brought(where, declared)) - except OSError as unreachable: - raise NotAFlow(f"{flow}: {unreachable}") from unreachable - - -def _brings(flow: str | os.PathLike[str]) -> tuple[str, ...]: - """The skills one flow named where it was declared, which live somewhere else. - - Args: - flow: The flow, as it was named -- the half after the colon says which of the flows in - the directory was asked for. - - Returns: - One identifier apiece, and nothing at all for a flow that named none. - """ - from hmz._legacy_flows import Flow as Marked - - from .finding import find, inside, loaded - - wanted = inside(str(flow)) - for one in loaded(find(str(flow))).values(): - said = getattr(one, "__humanize_flow__", None) - if isinstance(said, Marked) and said.name == wanted: - return said.skills - return () - - -def _called(flow: str | os.PathLike[str]) -> str: - """What a flow is called, given the file its entry point is in. - - Args: - flow: The path that was run. - - Returns: - The directory's name for a flow laid out as one -- the entry point is `__init__.py` in - every flow there is, and naming a flow after that would name them all the same -- and - the file's own name for a file somebody pointed at outright. - """ - from .finding import ENTRY - - said = Path(flow) - return said.parent.name if said.name == ENTRY else said.stem - - -def _carried( - flow: str, agents: Sequence[Agent], *, inherit: bool -) -> list[tuple[Loaded, ...]]: - """What each agent is to carry inside a called flow, worked out before the call takes it. - - Args: - flow: The flow being called, as it was asked for. - agents: The agents it is being handed, carrying the calling flow's own. - inherit: Whether what they carry now stays reachable inside the call, after the called - flow's own and only where the names do not collide. - - Returns: - One tuple apiece, in the order the agents were given. - """ - child = _brought(flow) - if child is None: - # A flow that says nothing about skills leaves them as they are, which is what a run - # of such a flow does too. - return [agent.loaded for agent in agents] - if not inherit: - return [child for _ in agents] - named = {one.name for one in child} - return [ - child + tuple(one for one in agent.loaded if one.name not in named) - for agent in agents - ] - - -class _Claim(NamedTuple): - """One call's hold on one agent, for as long as that call runs. - - Attributes: - on: The branch the call is on, which is what says whether two claims are one under the - other or two beside each other. - record: Where that call is written down, or None for a call nobody is keeping a record - of. - skills: What the called flow works by, which its agents carry while it runs. - config: What that call settled the agent to run as: the rung it may work at, its - goals and its searching. The person at the prompt's is what it already was, that - being a place a flow says none of these about. - """ - - on: Running - record: Sub | None - skills: tuple[Loaded, ...] - config: AgentConfig - - -class _Claimed(NamedTuple): - """One agent, as it was before any call took it and as the calls that have it want it. - - Attributes: - was: Where it was writing before the first of them took it. - carried: What it was carrying then. - ran: What it was set up as then, which is what it goes back to once every call that - took it has let go. - held: Every call that has it now, in the order they took it. - """ - - was: Journal | None - carried: tuple[Loaded, ...] - ran: AgentConfig - held: list[_Claim] - - -#: Which calls have each agent now, by the agent's own identity. An agent belongs to the run -#: rather than to any one flow of it, and a called flow points it at that call's own record -#: and that flow's own skills for as long as the call lasts -- so what it was before has to -#: be kept somewhere, and a swap remembered by whoever swapped it is a swap two calls going -#: at once put back in the wrong order. -#: -#: Read and written under `_TELLING`, beside what is running, since the two answer one -#: question: which flow this agent is working for now. -_CLAIMED: dict[int, _Claimed] = {} - -#: Where each flow of a run is writing, by the identity of its record. What a call made from -#: inside a flow is written under, which is the flow's own record and not the agents': an -#: agent two calls have at once is writing where both of them were called from, and a third -#: flow called from inside one of those belongs under that one. -_WRITTEN: dict[int, Sub | None] = {} - - -def _beneath(one: Running, of: Running) -> bool: - """Whether one flow is the other, or is running somewhere under it. - - Args: - one: The flow being asked about. - of: The flow it may be running under. - - Returns: - True where it is, which is what makes two claims on one agent a nesting rather than a - pair of siblings. - """ - at: Running | None = one - while at is not None: - if at is of: - return True - at = at.under - return False - - -def _points(agent: Agent, claimed: _Claimed) -> None: - """Points one agent at where the calls that have it agree it is working. - - One call has it: that call's record and that flow's skills, which is a called flow being - driven as a run of it would be. Two calls that are one under the other: the inner one's, - which is the same thing said twice. Two calls beside each other -- a flow that gathered - two calls sharing the agents it was handed -- and neither of them may have it: what a - session opened by that agent is part of is the flow they were both called from, and - writing it into whichever of them started last would be filing it under a flow that - happened to be there. A flow that wants a branch of its own writes down the agents for - it, which is what `drives` and `Agent.clone` are for. - - Which is the flow they were both called from and not the run: the fork may be five flows - down, and an agent dropped all the way back to the run's own record would be filed under - a flow that was not even in the room. So it is the deepest call holding this agent that - every one of them is running under -- and only where the fork is the run's own flow, which - holds nothing, does that come back to what the agent was before any of them took it. - - Args: - agent: The agent. - claimed: What it was, and what has it now. - """ - deep = sorted(claimed.held, key=lambda one: one.on.depth, reverse=True) - # The innermost, where the calls holding it are a chain -- a flow that called a flow. - if all(_beneath(deep[0].on, one.on) for one in claimed.held): - found = deep[0] - else: - # Otherwise the fork: the deepest of them that all of them are running under. - found = next( - ( - each - for each in deep - if all(_beneath(one.on, each.on) for one in claimed.held) - ), - None, - ) - if found is None: - agent.epic, skills = claimed.was, claimed.carried - runs = claimed.ran - else: - agent.epic, skills = found.record, found.skills - runs = found.config - _settles(agent).loads(skills) - if agent.config != runs: - # What it may do, whether it has goals and whether it reads the internet: the call - # holding it said them, and a call that has ended said them no longer. Nothing here - # can be refused -- every one of these was accepted on the way in. - _settles(agent).reconfigure(runs) - - -def _takes( - driven: Sequence[Agent], - started: Running, - record: Sub | None, - skills: Sequence[tuple[Loaded, ...]], - runs: Sequence[AgentConfig], -) -> None: - """Hands the agents to a call: its record to write into, its flow's skills to carry. - - Args: - driven: The agents the called flow was handed. - started: What :func:`entered` answered with. - record: Where the call is written down, or None for one nobody is keeping a record of. - skills: What each of the agents is to carry, in the order they were given. - runs: What each of them was set up as before this call, in the same order. - """ - with _TELLING: - _WRITTEN[id(started)] = record - for agent, carrying, was in zip(driven, skills, runs, strict=True): - held = _CLAIMED.setdefault( - id(agent), _Claimed(agent.epic, agent.loaded, was, []) - ) - held.held.append(_Claim(started, record, carrying, agent.config)) - _points(agent, held) - - -def _gives_back(driven: Sequence[Agent], started: Running) -> Sub | None: - """Hands the agents back as the call found them, and answers with the call's record. - - However the call ended, and whichever order two calls going at once end in: what an agent - goes back to is what it was before any call took it rather than what the call that is - ending happened to see, which two ending out of order would put back as each other's. - - Args: - driven: The agents the called flow was handed. - started: What :func:`entered` answered with. - - Returns: - Where the call was written down, or None for one nobody kept a record of. - """ - with _TELLING: - record = _WRITTEN.pop(id(started), None) - for agent in driven: - held = _CLAIMED.get(id(agent)) - if held is None: - continue - held.held[:] = [one for one in held.held if one.on is not started] - if held.held: - _points(agent, held) - continue - del _CLAIMED[id(agent)] - agent.epic = held.was - _settles(agent).loads(held.carried) - if agent.config != held.ran: - _settles(agent).reconfigure(held.ran) - return record - - -def _writes(driven: Sequence[Agent]) -> Epic | None: - """The record the flow running here writes to, which is what a call of its own goes under. - - Read off the branch this task is on rather than off the agents it is driving. An agent two - calls have at once is writing where both of them were called from, and a third flow called - from inside one of those belongs under that one rather than under what the two share. - - The flow nobody called keeps no record of its own here -- it writes the run's, which - whatever opened the run handed to the agents *it* started with. Those, and not the ones - this call is being made with: a branch driving agents of its own, which is what `drives` - and `Agent.clone` hand it, would otherwise be a branch that could not find the run it is - part of and would be written down nowhere, along with everything under it. - - Args: - driven: The agents the call is being made with, for a call from a thread with no branch - on it: one made from a tool a turn reached for, or from outside any flow at all. - - Returns: - The record, or None for a call from a flow nothing is keeping a record of -- one run - from a test, one called from nothing. - """ - from hmz.runtime.epic import Epic - - at = _ON.get() - if at is None: - # No branch to read: a call made from a thread that is not running a flow -- a tool - # a turn reached for, which is the flow's own code on somebody else's thread. What - # the agents are writing to now is then the only thing that knows. - return next( - (one for one in (each.epic for each in driven) if isinstance(one, Epic)), - None, - ) - over: Sequence[Agent] = driven - with _TELLING: - while at is not None: - if id(at) in _WRITTEN: - return _WRITTEN[id(at)] - if (held := _RUNNING.get(id(at))) is not None and held.agents: - over = held.agents - at = at.under - # What the agents of the outermost flow were handed as the run began -- and what they - # are still writing to unless a call has them, which is what was kept when it did. - was = [ - _CLAIMED[id(one)].was if id(one) in _CLAIMED else one.epic for one in over - ] - return next((one for one in was if isinstance(one, Epic)), None) - - -#: How deep one flow calling another goes before the next call is refused. A `load` chain has -#: no natural bottom -- a flow may call itself, and one that decides how deep to go from its -#: own config or from what a model said may decide wrong -- and what an unbounded one comes to -#: is a `RecursionError` out of whatever the innermost call happened to be importing, which -#: names no flow and blames the wrong line. High enough that no chain anybody writes on -#: purpose reaches it, low enough to be reached long before the interpreter's own limit is. -_DEEPEST = 64 - - -def _deep(flow: str) -> None: - """Refuses a call from a chain of flows that has gone deeper than one goes. - - Args: - flow: The flow being called, as it was asked for. - - Raises: - NotAFlow: If calling it would be deeper than :data:`_DEEPEST` flows down. - """ - at = _ON.get() - if at is None or at.depth + 1 <= _DEEPEST: - return - walked = " > ".join(one.flow for one in running()[-3:]) - raise NotAFlow( - f"{flow}: called {at.depth + 1} flows deep, and a chain of flows calling flows " - f"goes {_DEEPEST} -- a flow with no bottom to it is a flow to correct, and this " - f"one reached here through … > {walked}" - ) - - -def load(flow: str | os.PathLike[str], *, inherit_skills: bool = False) -> Entry: - """One flow, ready for another flow to run: what it marked, found by name. - - A flow is a loop over agents, and a loop worth having is one another loop can reach for:: - - from hmz._legacy_flows import Agent, flow, load - - @flow - def run(agents: tuple[Agent, Agent], task: str) -> None: - plan = load("official/humanize1:gen-plan") - plan(agents, f"plan this first: {task}") - agents[0].new()(task) - - The name is the one `-f` takes -- `ralph_loop`, `official/rlar`, `humanize1:gen-plan`, a - path of your own -- so a flow reaches another flow the way a person does, and a flowverse - is a library as well as a menu. - - Loading rather than calling, because that is what this does: what comes back is a flow to - run, and running it is the caller's own line. It is not :func:`hmz._legacy_flows.loaded`, - which is - what running a flow's file leaves behind -- one loads a flow, the other reads a file. - - What comes back is the flow's own function, with the run written down around it: what is - running is what the interface shows, and a flow that called another must not read as the - flow that was started. It is called the way the flow itself is -- the agents, the task, - and the config for one that says it takes one -- and answers with whatever the flow - answers with, so a flow written as a coroutine is awaited by whoever called it:: - - await load("official/rlar")(agents, task) - - The flow is read again at each call, and so are the skills it brings. A flow is a - directory on disk, and one that has been rewritten between two calls of it -- by hand, or - by an agent this very flow is driving -- is run as it is now rather than as it was when - somebody first asked for it. That is what makes a loop that improves its own flow, or its - own skills, a loop that then runs the improved one. - - A called flow carries only its own skills by default. A wrapper flow may explicitly pass - its skills through with ``inherit_skills=True``. The called flow still owns the result: - its skill wins when parent and child use the same name, and the agents are restored to - exactly what the caller carried when the call returns or raises. - - A call may also say what the flow it is calling runs at, which is `drives`: a mapping - from the name the called flow gives one of its places -- or the name of the agent filling - it -- to the config that branch is to be driven at:: - - await asyncio.gather( - load("official/rlar")(agents, task), - load("official/rlar")(agents, task, drives={"actor": careful}), - ) - - What each of those is handed is a clone at that config rather than the agent set up - again: an agent is what it was made as, so two efforts are two agents. They are the - call's own -- written into the call's own record, carrying the called flow's skills -- - which is also what makes two calls gathered at once two branches with nothing shared - between them. - - Each call is written down as the run of a flow it is. The record of the flow that called - it gets a line saying so -- one record per call, named for the flow and for this call of - it -- and what the called flow opens, keeps and calls in turn goes there rather than into - the record of whatever started the run. So a flow calling a flow calling a flow reads - back as the tree it ran as, however deep it went and however many of it ran at once. The - record that called it says `called` and `returned` with the filename, at both ends, - because two calls going at once end in an order nothing can pair by. - - Args: - flow: The flow to call, by the name `-f` takes. - inherit_skills: Whether skills carried by the calling flow remain available inside the - called flow, after the called flow's own and only where their names do not collide. - - Returns: - Something to call with the agents and the task. - - Raises: - NotAFlow: If there is no such flow, or it is not one. Raised here rather than at the - call, so that a flow which asks for another by a name that is wrong says so when it is - asked for rather than an hour into a loop -- and again at each call, for a flow that - was rewritten into something that is no longer one. - """ - # Said now, so a name that is wrong -- or an atlas whose body will not compile -- is - # wrong where it was written rather than an hour into a loop. - readies(declares(flow)[0]) - named = str(flow) - - def calling( - agents: Sequence[Agent], - task: str, - config: BaseModel | dict[str, Any] | None = None, - *, - drives: Mapping[str, AgentConfig] | None = None, - ) -> Awaitable[None] | None: - # Read afresh, which is what makes a flow rewritten since the last call the flow that - # runs now: a flow is a directory, and reading one is running its entry point. - run, places, make, setting, mark = declares(flow) - driven = _handed(named, places, make, agents, drives) - # Read back through the flow's own model, which is what refuses a config a flow does - # not take and one it takes another of -- and what puts the settings through its own - # validators at the moment it is about to run, exactly as a run of it does. Before the - # skills below, because a refusal here is a call that never happened: a caller that - # catches it -- to try another config, or to go on without this flow -- must not be - # left driving agents that are carrying the skills of a flow that never ran. - given = None if config is None else set_up(named, setting, config) - settings = () if setting is None else (given,) - # And refused where a chain of them has no bottom, for the same reason and in the - # same breath: before anything has been taken, carried or written down. - _deep(named) - if _awaits(run): - # Nothing is taken until the flow itself starts. A coroutine has not run when it - # is made -- one gathered and then cancelled before its first step never runs at - # all -- so a call written down as started by the making of it would be a call - # nothing ever ends, holding agents nothing ever hands back. It is also where - # the branch has to be taken: `asyncio` copies the context into the task that - # runs it, and two gathered at once are two tasks with a context apiece. - return _running( - run, - driven, - named, - places, - task, - settings, - resumable=mark.resumable, - inherit=inherit_skills, - ) - started, held = _begins( - named, - places, - driven, - task, - resumable=mark.resumable, - inherit=inherit_skills, - ) - try: - answered = run(driven, task, *settings, *held) - except BaseException as why: - _ended(driven, started, type(why)) - raise - if inspect.isawaitable(answered): - # A flow that is not a coroutine function and answered with something to await - # all the same -- one wrapped in a decorator of its own, a compiled atlas. It - # runs while whoever called it awaits it, so what says it is running has to last - # that long too, and the branch goes back to the caller until it does: here is - # the caller, and two of these gathered at once share it. - _ON.set(started.under) - return _awaited(answered, driven, started) - _ended(driven, started) - return None - - return calling - - -def _awaits(run: Entry) -> bool: - """Whether a flow is one that has to be awaited, asked before it is called rather than after. - - Asked beforehand because a coroutine flow must be written down as started where it starts - rather than where it was made, and what it was made by is the only thing there is to ask - at that point. A flow that is not one of these and answers with something to await anyway - -- one wrapped in a decorator of its own -- is left to be found out by the answer. - - Args: - run: The flow's entry point. - - Returns: - True where calling it gives back a coroutine. - """ - if inspect.iscoroutinefunction(run): - return True - # A flow that is an object rather than a function, which is what a compiled atlas is. - called = getattr(run, "__call__", None) # noqa: B004 -- asked of it, not called - return called is not None and inspect.iscoroutinefunction(called) - - -def _begins( - named: str, - places: tuple[Place, ...], - driven: tuple[Agent, ...], - task: str, - *, - resumable: bool, - inherit: bool, -) -> tuple[Running, tuple[Any, ...]]: - """Puts a call on the branch it runs on, and takes the agents for it. - - Args: - named: The flow being called, as it was asked for. - places: What it declared, which is what says how its agents run while it has them. - driven: The agents it is being handed. - task: What it was called with. - resumable: Whether it says it can be picked up again. - inherit: Whether the calling flow's skills stay reachable inside it. - - Returns: - What :func:`entered` answered with, and what the flow is to be called with after the - task and its settings. - - Raises: - NotAFlow: If one of the agents cannot be told what the flow says its place runs at. - """ - # Where it is written down, which is under the flow running here rather than under - # whatever the agents happen to be writing to: two calls sharing one agent leave it - # writing where they were both called from, and a third called from inside one of them - # belongs under that one. - under = _writes(driven) - # What it left behind last time, for a flow that says it can be picked up: kept under its - # own name in the epic of the run that called it, since a flow that called another is two - # flows and neither writes the other's. - held = () if not resumable else (_holding(under, named),) - # And the skills it works by, which are the flow's rather than the agents': a called flow - # brings its own, mounted onto whatever sessions it opens, and hands the agents back as - # it found them so that the flow which called it goes on carrying its own. - carrying = _carried(named, driven, inherit=inherit) - started = entered(named, driven) - try: - # A record of its own to write into, in the epic of the run that called it: a called - # flow opens sessions and calls flows of its own, and what it did is its own rather - # than a run's that happened to start it. - writing = _opened(under, driven, named, task, resumable=resumable) - # And what it says its agents may do, whether they have goals and whether they read - # the internet, which are the called flow's for the length of the call: they are - # handed back as they came, as they are with the skills and the record. - were = _settled(named, places, driven) - _takes(driven, started, writing, carrying, were) - except BaseException: - # A record that could not be opened is a call that never started: leaving it on the - # branch would put every later call of this task under a flow that is not running. - left(started) - raise - return started, held - - -async def _running( - run: Entry, - driven: tuple[Agent, ...], - named: str, - places: tuple[Place, ...], - task: str, - settings: tuple[Any, ...], - *, - resumable: bool, - inherit: bool, -) -> None: - """Runs a flow written as a coroutine, taking its agents where it actually starts. - - Args: - run: The flow's entry point. - driven: The agents it was called with. - named: The flow, as it was asked for. - places: What it declared. - task: What it was called with. - settings: What it was set up with, or nothing for a flow that takes none. - resumable: Whether it says it can be picked up again. - inherit: Whether the calling flow's skills stay reachable inside it. - """ - started, held = _begins( - named, places, driven, task, resumable=resumable, inherit=inherit - ) - try: - # Asked of the answer all the same: what says it has to be awaited is what it was - # written as, and a flow is what it does when it is called. - if inspect.isawaitable(answered := run(driven, task, *settings, *held)): - await answered - except BaseException as why: - # Cancellation among them: a task taken down mid-await unwinds through here, which - # hands its agents back, closes its record and takes it off the branch -- and does so - # at every level, each level being a task or a frame of its own. - _ended(driven, started, type(why)) - raise - _ended(driven, started) - - -async def _awaited( - answered: Awaitable[None], - driven: tuple[Agent, ...], - started: Running, -) -> None: - """Waits for a flow that answered with something to await, and writes down that it ended. - - The branch is taken here rather than where the call was written, for the reason a - coroutine flow's is: it is here that the flow is actually running, and here is a task of - its own where a sibling was gathered beside it. - - Args: - answered: What calling it gave back. - driven: The agents it was called with. - started: What :func:`entered` answered with. - """ - _ON.set(started) - try: - await answered - except BaseException as why: - _ended(driven, started, type(why)) - raise - _ended(driven, started) - - -def _opened( - under: Epic | None, - driven: tuple[Agent, ...], - named: str, - task: str, - *, - resumable: bool, -) -> Sub | None: - """Opens the record a called flow is written to, inside the record that called it. - - Args: - under: The record of the flow making the call, or None for a call from a flow nobody is - keeping a record of -- one run from a test, one called from nothing. - driven: The agents the called flow was handed. - named: The flow, as it was asked for. - task: What it was called with. - resumable: Whether it says it can be picked up again. - - Returns: - The record, or None where there is nowhere to write. - """ - if under is None: - return None - # Cast because a flow sees its agents through `Agent`, which says what a flow may ask - # of one and nothing about what it was configured with -- and what a record says it - # was driven by is exactly that. They are the run's own agents either way. - return under.called( - named, cast("Sequence[AgentBase]", driven), task, resumable=resumable - ) - - -def _ended( - driven: tuple[Agent, ...], - started: Running, - kind: type[BaseException] | None = None, -) -> None: - """Writes down that a called flow has ended, and hands its agents back as they came. - - Args: - driven: The agents it was called with. - started: What :func:`entered` answered with. - kind: What was raised out of the called flow, if anything. - """ - if (record := _gives_back(driven, started)) is not None: - record.ended(kind) - left(started) - - -def _handed( - flow: str, - places: tuple[Place, ...], - make: Callable[..., tuple[Agent, ...]], - agents: Sequence[Agent], - drives: Mapping[str, AgentConfig] | None = None, -) -> tuple[Agent, ...]: - """The agents a called flow is handed, as the tuple that flow declared. - - A flow is called with what it drives, so a caller hands over as many agents as the flow - declares -- and may hand over one fewer where the flow talks to the person, since the - person is made rather than chosen. Nothing is renamed: the agents belong to the flow that - was started, and a name changed under it would change what the run has already been - written down as. - - A caller may say what a place of the called flow is to be driven at, and what fills that - place is then a clone at that config -- a second agent rather than this one set up again, - since what an agent is is settled where it is made. It is checked exactly as the agent it - replaces would have been: a config that puts a place somewhere the flow does not is - refused where the call was written. - - Args: - flow: The flow being called, for what a refusal says. - places: What it declared. - make: What to build its agents as -- the named tuple it declared, or a plain one. - agents: What the caller handed over. - drives: What to drive one or more of its places at, by the name the called flow gives - the place or the name of the agent filling it, or None to drive them all as they come. - - Returns: - The agents, as the flow declared them. - - Raises: - NotAFlow: If that is the wrong number of them, if one of them cannot run a moment the - flow says that place has to, if one of them does not serve what the flow says that - place needs, if one is somewhere the flow does not put it, or if `drives` names - something the flow does not drive. - """ - from hmz.coganchor.agents import HumanAgent - - given = list(agents) - asked = [place for place in places if not place.person] - if len(given) == len(places): - driven = given - elif len(given) == len(asked): - # The person is made rather than chosen, exactly as a run of the flow makes one. - taking = iter(given) - driven = [HumanAgent() if place.person else next(taking) for place in places] - else: - raise NotAFlow( - f"{flow}: the flow drives {len(asked)} agents, {len(given)} given" - ) - if drives: - driven = _differently(flow, places, driven, drives) - for agent, place in zip(driven, places, strict=True): - if short := place.moments - type(agent).moments: - raise NotAFlow( - f"{flow}: {place.name or 'the agent'} has to run " - f"{', '.join(sorted(short))}, which {agent.backend} does not" - ) - # The same as a run of this flow asks, and asked here for the same reason: a place - # run under a goal, filled by an agent that has no goal feature or has had it - # switched off, is a call that fails at its first `pursue` -- hours in, from inside - # the called flow, rather than where the call was written. - if place.goal and not type(agent).pursues: - raise NotAFlow( - f"{flow}: {place.name or 'the agent'} is run under a goal, which " - f"{agent.backend} has no feature for" - ) - if place.goal and not agent.goals_enabled: - raise NotAFlow( - f"{flow}: {place.name or 'the agent'} is run under a goal, but goals " - "were switched off for it" - ) - serves(flow, agent, place) - lands(flow, agent, place) - return make(driven) - - -def _settled( - flow: str, - places: tuple[Place, ...], - driven: Sequence[Agent], -) -> tuple[AgentConfig, ...]: - """Sets every agent of a called flow up as that flow says its places run. - - Args: - flow: The flow being called, for what a refusal says. - places: What it declared. - driven: The agents it is being called with, in the order it declared them. - - Returns: - What each of them was set up as before this, in the same order, to be handed back when - the call ends -- the person among them left alone, who takes no turn a rung means - anything about. - - Raises: - NotAFlow: If one of them cannot be told what the flow says its place runs at. The ones - settled before it are put back first: a call that never happened must leave the flow - that tried it driving the agents it had, exactly as a refused config does. - """ - were: list[AgentConfig] = [] - dropped: list[str] = [] - try: - for agent, place in zip(driven, places, strict=True): - were.append( - agent.config - if place.person - else runs_at(flow, agent, place, dropped=dropped) - ) - except NotAFlow: - for agent, was in zip(driven, were, strict=False): - if agent.config != was: - _settles(agent).reconfigure(was) - raise - _aside(dropped, driven) - return tuple(were) - - -def _aside(said: Sequence[str], driven: Sequence[Agent]) -> None: - """Says what a called flow could not carry, where a call is the middle of a run. - - The run is already going by the time a flow calls another, so there is no `Runner` left - to ask and nowhere to put the answer until somebody asks for it -- these agents are - already being watched or already writing to a terminal. So it is said the way a driver - says a thing mid-run: on stderr, and only where nothing is watching. Whatever is watching - owns the screen, and it is drawing these agents' own lines onto it. - - Args: - said: The lines, which is empty for the calls that carried everything. - driven: The agents of the call, for whether anything is watching them. - """ - import sys - - from hmz.coganchor.agents.event import say - - if not said or any(cast("AgentBase", agent).watched for agent in driven): - return - for line in said: - say(line, sys.stderr) - - -def _differently( - flow: str, - places: tuple[Place, ...], - driven: Sequence[Agent], - drives: Mapping[str, AgentConfig], -) -> list[Agent]: - """The agents of a call that said what one of its places is to be driven at. - - A clone apiece rather than the agents set up again, because that is what an agent set up - differently is: an agent is what it was made as, and two efforts are two agents. Each - carries what the one it stands in for carries and is named nothing, which is what tells - a comparison of two efforts from one agent that changed its mind. - - Args: - flow: The flow being called, for what a refusal says. - places: What it declared, which is what its places are called. - driven: The agents it would have been handed. - drives: What to drive one or more of those places at. - - Returns: - The agents to hand over, the clones among them. - - Raises: - NotAFlow: If it names something the flow does not drive, something two of its places - answer to, or the person at the prompt -- who runs nothing anybody chose and so has - nothing to be driven at. - """ - from hmz.coganchor.agents import HumanAgent - - made = list(driven) - where: dict[str, int | None] = {} - for at, agent in enumerate(made): - # None for a name two places answer to -- one agent handed to a flow twice, two named - # the same -- since what a caller meant by it is then not a thing to guess at. - where[agent.id] = None if agent.id in where else at - # The called flow's own names over the agents' own: a caller says what the flow it is - # calling is to drive its reviewer at, and what the flow it was handed calls that agent - # is the caller's business rather than the callee's. - for at, place in enumerate(places): - if place.name: - where[place.name] = at - for name, config in drives.items(): - if name in where and where[name] is None: - raise NotAFlow( - f"{flow}: two of the agents it is being handed are called {name!r}, so " - "which of them is to be driven at that is not a thing to work out" - ) - at = where.get(name) - if at is None: - raise NotAFlow( - f"{flow}: nothing it drives is called {name!r} -- it drives " - f"{', '.join(place.name or 'an agent' for place in places)}" - ) - if isinstance(made[at], HumanAgent): - raise NotAFlow( - f"{flow}: {name} is the person at the prompt, who takes no turn anywhere " - "and so runs nothing to be driven at" - ) - made[at] = made[at].clone(config=config) - return made - - -def _holding(under: Epic | None, named: str) -> dict[str, Any]: - """The dict a called flow that can be picked up writes what it wants back into. - - Kept in the epic of the run that called it, under the called flow's own name: a flow - that called another is two flows, each with its own to keep, and both of them part of one - run. A call from a flow that opened no epic -- one run from a test, one called from - nothing -- is handed a dict that is nowhere, which is a flow that runs and leaves nothing - rather than a call that fails. - - Found through the flow making the call rather than through the agents it is handing over, - for the reason the record is: a branch driving agents of its own would otherwise leave - nothing behind and be picked up as a run that never happened. - - Args: - under: The record the flow making the call is writing, which is what holds the epic. - named: The called flow, as it was asked for. - - Returns: - What it left behind last time, as something to write this time's into. - """ - from hmz.runtime.epic import resumed, state - - if under is None: - return {} - at = resumed(named, under.workspace) - return under.state(named, state(at, named) if at is not None else None) - - -def serves(flow: str | os.PathLike[str], agent: Agent, place: Place) -> None: - """Refuses an agent whose backend does not serve what the flow says its place takes. - - Before the first turn, for the reason a moment the place hangs a hook on is checked - before it: a flow built on a turn that can be talked to while it runs, or on one held to - a shape, finds out from the call that reached for it otherwise -- hours into a loop, - rather than from the line that chose the agent. So it is asked of the driver class and of - what is written down about the CLI, neither of which needs an agent to have opened - anything, and a flow that cannot be driven by what it was handed says so at once. - - Args: - flow: The flow, for what a refusal says. - agent: The agent filling the place. - place: What the flow declared. - - Raises: - NotAFlow: If the backend does not serve something the place says it has to, or if what - it asked of the backend is something only a machine can answer -- `Needs("isolated")` - rather than `Needs(where=("isolated",))`. That one used to pass whatever was filling - the place: a machine capability carries no backends, an empty backend set means every - backend, and so a flow that wrote the ask in the wrong half was told nothing and - protected by nothing. - """ - if place.needs is None or not place.needs.of_agent: - return - from .checking import INSIDE, OF_AGENT, misplaced - - called = place.name or "the agent" - if wrong := misplaced(place.needs.of_agent, OF_AGENT): - raise NotAFlow( - f"{flow}: {called} asks for {', '.join(sorted(wrong))} of the agent, which " - "is asked of where it works instead -- " - f"Needs(where={tuple(sorted(wrong))!r})" - ) - served = comes_to(agent.backend) - if agent.config.machine is not None: - # Reaching inside the CLI is something humanize does to a process it started here, - # and a turn whose process is on another machine is one none of that reached. The - # drivers each switch it off for that reason, so a place that asked for it is refused - # here rather than given the weaker thing without a word. - served -= INSIDE - if short := place.needs.of_agent - served: - where = " where its turns land" if short & INSIDE else "" - raise NotAFlow( - f"{flow}: {called} has to serve " - f"{', '.join(sorted(short))}, which {agent.backend} does not{where}" - ) - - -def comes_to( - backend: str, *, catalogued: Sequence[Capability] | None = None -) -> frozenset[str]: - """What one backend serves, by the names a flow asks for it under. - - Read out of the one catalogue rather than off the driver classes again: what a flow may - build on -- the shape a turn can be held to, the tools it may be offered, the word put - into a turn already running, the goal feature, the fork, each moment outside the ones - every backend reaches -- is already read off those classes there, and a second reading - here would be a second place for it to be wrong. What the catalogue leaves to the - backend's own facts is read where those are written down, which is the same profile it - reads. - - By name rather than by agent, so that whoever is *choosing* an agent can ask before there - is one: the picker rules a CLI out for a place it could not fill, and a run refuses one - that was handed over anyway, and both are asking this one question. - - Args: - backend: The coding agent, named as a command line names it. - catalogued: The catalogue to read, for a caller asking this of several backends at - once -- a picker listing every CLI a place could be filled by asks twelve times, and - the catalogue is read off the live interface with `inspect` each time it is built. - None reads it here, which is what one question wants. - - Returns: - The names it serves, out of the half of the vocabulary a backend answers for. What every - backend here serves is in it too: the catalogue names no backend against those because - they are true of all of them, and a place that asked for one would otherwise be refused - every agent there is. - - What a machine comes to is not in it and must not be. Those capabilities carry no - backends for the same reason -- no backend answers for them -- and reading that as "all - of them" is what made `Needs("isolated")` a check that measured nothing. - """ - from hmz.coganchor.backends import named - - from .checking import OF_AGENT, catalogue - - held = catalogue() if catalogued is None else catalogued - comes = { - one.name - for one in held - if one.asked == OF_AGENT and (not one.backends or backend in one.backends) - } - profile = named(backend) - return frozenset(comes if profile is None else comes | profile.tags()) - - -def lands( - flow: str | os.PathLike[str], - agent: Agent, - place: Place, - *, - container: str = "", -) -> None: - """Settles where one agent's turns land, and refuses a machine the flow did not allow. - - Where an agent works is the flow's to say and not a setting anybody may reach for: a flow - is written for one shape of work, and one whose agents read this project cannot have one - of them reading somebody else's. So a place says nothing and its agent runs here, or says - `Remote` and its agent may be pointed at a machine by whoever chose it, or says `Isolated` - and the machine is the flow's own -- a container of the image it named, which nobody else - has any say in. - - A whole run put in a container from outside is none of those three: it was said once, - about every agent, by whoever started the run, and an agent standing in it was pointed - nowhere by anybody. So a place that says nothing takes one, and goes on refusing the - machine somebody actually chose for it. A place that says `Isolated` does not: that flow - named an image of its own, and being handed the run's container instead is being pointed - at a machine, which is what it says nobody may do. - - Args: - flow: The flow, for what a refusal says. - agent: The agent filling the place. - place: What the flow declared. - container: The image the whole run works in, for a run put in one from outside, or "" - for a run on this machine. Named rather than read off the agent because the container - is started where the run starts and nothing is pointed at it until then: a place that - needs somewhere remote would otherwise be refused at the top of a run that is about - to put every agent of it somewhere remote, and allowed inside that same run when a - flow called another. It is one question, so it is asked of one answer. - - Raises: - NotAFlow: If the agent was configured to work somewhere the flow does not put it, if - where it works does not come to what the flow says that place needs, or if it has - already opened a session, which is a conversation that cannot be moved. - """ - from hmz.coganchor.agents import Isolated, isolated - - called = place.name or "the agent" - if isinstance(place.where, Isolated): - if agent.config.machine is not None: - raise NotAFlow( - f"{flow}: {called} works in a container of this flow's own, so there is " - "nothing to point it at" - ) - # Against the container this flow named, and before the agent is put in it: a call - # that is refused must leave the flow which made it driving the agents it had, and - # one whose place had already been moved would hand back an agent pointed somewhere. - box = isolated(place.where.image) - _somewhere(flow, called, box, place) - try: - _settles(agent).runs_on(box) - except RuntimeError as opened: - raise NotAFlow(f"{flow}: {called} {opened}") from opened - return - # The container the whole run works in, which is a convenience rather than a second way - # of saying where an agent works -- so a flow this one called must not read it as one. - # By identity, since what is exempt is that container and not the idea of a machine. - inside = bool(_INSIDE) and agent.config.machine is _INSIDE[0][1] - if place.where is None and agent.config.machine is not None and not inside: - raise NotAFlow( - f"{flow}: {called} runs on this machine -- this flow does not say it works " - "anywhere else, so it cannot be pointed at one" - ) - _somewhere(flow, called, agent.config.machine or _run_in(container), place) - - -def _run_in(image: str) -> MachineConfig | None: - """The machine a run put in a container from outside will be working in. - - Built rather than started: what a container comes to is its settings' to answer, and - reading a flow must not pull an image. - - Args: - image: The image the whole run works in, or "" for a run on this machine. - - Returns: - The settings every agent of that run will be pointed at, or None for a run here. - """ - from hmz.coganchor.agents import isolated - - return isolated(image) if image else None - - -def _somewhere( - flow: str | os.PathLike[str], - called: str, - machine: MachineConfig | None, - place: Place, -) -> None: - """Refuses a place whose machine does not come to what the flow says the work takes. - - Asked of the machine's settings and never of a machine, which is the whole reason those - settings answer it: a flow whose work has to happen somewhere isolated must be refusable - before an image has been pulled, and one whose work has to happen on Linux before a - connection has been made. What only the machine itself can settle it does not claim -- - the platform of somebody else's machine is read from the handshake, and a machine that - turns out not to be what its settings promised fails as it starts -- so nothing here - waits on anything being up. - - Args: - flow: The flow, for what a refusal says. - called: What the flow calls the place. - machine: The settings the work would land under, or None for this machine. - place: What the flow declared. - - Raises: - NotAFlow: If those settings do not come to something the place says it needs, or if what - it asked of the machine is something only the agent can answer. An agent pointed - nowhere works on this machine, which comes to nothing at all, so a place that needs - anything of where it works needs a machine first. - - `Needs(where=("anchor:hooked",))` is the second of those and used to be advertised: - a hook table and a preload are the CLI's own to take, no machine's settings have ever - carried either, and so that ask was refused by every machine there is -- a check - nothing could pass, which is worse than no check. It is asked of the agent, where the - profile that declares it can answer, and where `serves` takes it away again from an - agent whose turns land somewhere humanize cannot reach into. - """ - if place.needs is None or not place.needs.where: - return - from .checking import WHERE, misplaced - - if wrong := misplaced(place.needs.where, WHERE): - raise NotAFlow( - f"{flow}: {called} asks for {', '.join(sorted(wrong))} of where it works, " - "which is asked of the agent instead -- " - f"Needs({', '.join(repr(one) for one in sorted(wrong))})" - ) - at: frozenset[str] = machine.capabilities if machine is not None else frozenset() - if short := place.needs.where - at: - raise NotAFlow( - f"{flow}: {called} has to work somewhere that comes to " - f"{', '.join(sorted(short))}, which " - f"{'the machine it works on' if machine is not None else 'this machine'} " - "does not" - ) - - -#: What a place may have an opinion about, and the answer each of them has where the place has -#: none. Read two ways in one line of :func:`runs_at`: to build what the place asks for, and to -#: work out which of the settings a refusal names were the place's to give up. -_DECLARABLE: Final = ("permission", "goals", "web_search") - - -def _declared(place: Place) -> frozenset[str]: - """Which of the three the place actually raised the subject of. - - A place declaring nothing declares nothing at all, and the fields carry that as three - different silences -- `UNSAID` is the empty rung, None is no answer about the web, and - `goals=True` is the goal feature left wherever it was. Each of them settles nothing when - it meets an agent, so each of them is nothing to drop either. - - Args: - place: What the flow declared. - - Returns: - The `AgentConfig` field names the place gave an answer for, which is what leniency is - allowed to take back and the whole of it. A tier the agent was built with and an effort - somebody typed are refused whatever the place says about insisting: the place never - asked for them, so dropping them would be dropping somebody else's answer. - """ - said = { - "permission": bool(place.permission), - "goals": place.goal or not place.goals, - "web_search": place.web_search is not None, - } - return frozenset(field for field in _DECLARABLE if said[field]) - - -def _instead(setting: str, was: AgentConfig) -> str: - """What the agent will actually do about a setting that was dropped, in words. - - The whole point of saying a setting was dropped is saying what stands in its place, and - what stands in its place is whatever the agent was already carrying. Worded rather than - printed as a value, because the value that matters most here is a silence: `None` and - `""` are the two answers that read as nothing on a screen and mean something particular. - - Args: - setting: The `AgentConfig` field. - was: The agent as it was before the place was settled onto it. - - Returns: - The half-sentence that follows "so". - """ - kept = getattr(was, setting, None) - if setting == "web_search": - if kept is None: - return ( - "it goes on reading the web exactly as whoever installed its CLI has it" - ) - return f"it runs with web_search={kept!r}" - if setting == "permission": - if not kept: - return "it runs at whatever rung its own CLI's headless run leaves it at" - return f"it runs at {kept!r}" - return f"it runs with {setting}={kept!r}" - - -def _dropped( - flow: str | os.PathLike[str], - place: Place, - agent: Agent, - was: AgentConfig, - settings: Iterable[str], -) -> str: - """The line that says a declaration was not carried, so that nothing is carried quietly. - - Args: - flow: The flow, said the way a refusal says it. - place: What the flow declared. - agent: The agent filling it. - was: What that agent carries, which is what it goes on carrying. - settings: The `AgentConfig` fields being given up. - - Returns: - One line, naming the place, the setting, and what the agent will do instead. - """ - each = ", ".join( - f"{setting} ({_instead(setting, was)})" for setting in sorted(settings) - ) - return ( - f"{flow}: {place.name or 'the agent'} declares what {agent.backend} cannot be told, " - f"so that declaration is dropped rather than kept as one that lies -- {each}. " - "The place did not insist; `AgentDefaults(insist=True)` beside it refuses the run " - "instead." - ) - - -def runs_at( - flow: str | os.PathLike[str], - agent: Agent, - place: Place, - *, - dropped: list[str] | None = None, -) -> AgentConfig: - """Settles what one agent may do, whether it has goals and whether it reads the internet. - - The three things a flow says about the work rather than about the agent, and the flow is - the only one that says them: whoever chose the agent chose a CLI, a model, an effort and - an account, and a rung typed there would be somebody outside the flow deciding what the - flow's reviewer is allowed to rewrite. So a place carries them, and this is where they - reach the agent -- before its first turn, over whatever it was constructed with. - - Tighter only, never looser. A place that declares nothing declares nothing at all -- no - rung, no answer about the web, goals left as they were -- and what an agent already - carries is never loosened to reach it, so a flow declaring nothing runs its agents at - exactly what they came with. It is the same rule that makes a call safe: a flow running at - `read-only` that called one which declared nothing would otherwise run that one at no rung - at all -- looser than any rung, since nothing said is the CLI left to decide -- and calling - a flow somebody else wrote would be how a person's `read-only` gets undone. - - Tighter only is one axis, and how hard the declaration binds is the other. A place that - writes `insist=False` says it would rather run on a backend that cannot be told than not - run at all -- which is the honest thing for a benchmark to want, `web_search=False` being - a statement about the comparison rather than about the roster, and three backends with no - switch being three noisier cells rather than three cells that never happened. Leniency - then does exactly one thing, and it is not ignoring the setting: the setting that could - not be carried is **dropped**, the rest are settled, and what was dropped is said out - loud, naming the place, the setting, and what the agent does instead. A config left - carrying `web_search=False` on an agent that will go on searching is the failure this - whole seam exists to prevent, and it is no less a failure for having been asked for - politely. - - Only what the place declared may be dropped. A refusal naming `service_tier` or a rung of - `effort` is a refusal about what the agent was built with -- nobody in the flow asked for - it, so nobody in the flow may give it up -- and it is raised whatever the place says about - insisting. That is what the typed refusal is for: a sentence cannot be asked which field - it was about, and this needs to ask. - - Settled in more than one pass, because :meth:`_serves` raises at the first thing it finds - and there is no way to ask a backend for the whole list. A place declaring two settings - against a backend that can carry neither is refused once about the first, drops it, and is - refused again about the second -- so each pass gives up at least one field, and there are - only three fields, so it ends. Tried again rather than worked out ahead of time because - what a backend serves is the backend's to say: the alternative is a table here of what - every driver refuses, which is the same table twice and the second copy always wrong. - - Args: - flow: The flow, for what a refusal says. - agent: The agent filling the place. - place: What the flow declared. - dropped: Where to put a line about each declaration that was not carried, or None for - a caller with nowhere to put one. Collected rather than printed, for the reason - `Runner.unreadable` is: the object that knows answers, and whoever has a screen does - the saying -- `hmz exec` on stderr, the interface in the transcript. - - Returns: - What the agent was set up as before this, so that a flow which called another can hand - it back exactly as it found it. - - Raises: - NotAFlow: If the backend has no way of being told what the flow said -- a CLI that - cannot be told not to search the web is a CLI that would go on searching, which is a - declaration that lies rather than one that holds. Unless the place wrote - `insist=False` and the refusal is about something the place itself declared, in which - case that one declaration is given up instead and a line about it goes to `dropped`. - """ - from dataclasses import replace - - from hmz.coganchor.agents import Unserved, searching, tightest - - was = agent.config - declared = _declared(place) - giving: frozenset[str] = frozenset() - while True: - try: - # Inside the try because writing the config is where a backend refuses as surely - # as settling it is: a config of its own may hold two of these settings against - # each other -- a rung it can only say in a table this agent was told not to - # write -- and that refusal is the same refusal, owed the same sentence about - # which place it was. - asks = { - "permission": tightest(was.permission, place.permission), - # A place run under a goal has one whatever the agent came with: an agent - # with goals switched off is refused where the place is filled rather than - # quietly run without. - "goals": place.goals if place.goal else (was.goals and place.goals), - "web_search": searching(was.web_search, place.web_search), - } - # A setting given up is the place's answer taken back out, which leaves the - # agent's own -- not some third value, and not the place's applied quietly. - wanted = replace( - was, - **{ - field: getattr(was, field) if field in giving else asks[field] - for field in _DECLARABLE - }, - ) - if wanted == was: - break - _settles(agent).reconfigure(wanted) - except Unserved as refused: - give = refused.settings - giving - if not place.insist and give and refused.settings <= declared: - giving |= give - continue - raise NotAFlow( - f"{flow}: {place.name or 'the agent'} cannot be run as this flow declares " - f"-- {refused}" - ) from refused - except ValueError as refused: - raise NotAFlow( - f"{flow}: {place.name or 'the agent'} cannot be run as this flow declares " - f"-- {refused}" - ) from refused - break - if giving and dropped is not None: - dropped.append(_dropped(flow, place, agent, was, giving)) - return was - - -def _unfetched(named: str) -> str: - """Why a flow that was named is not there, as far as that can be told. - - Args: - named: What was asked for, as it was written. - - Returns: - The reason: that a flowverse it could have come from has not been fetched yet, where - that is what happened, and otherwise that there is no such file. A flowverse is offered - before it is fetched -- `official` is there from the start -- so "no such file" would be - the answer to a name that is right, given by the one thing that knows it has not been - downloaded. - - A name that said which place it came from is a question about that place alone. A bare - one is looked for in every one of them, and humanize's own flows are said by a bare name - now, so the first run on a machine that has fetched nothing is exactly where one of them - goes missing -- and "no such file" is the least useful thing to say about it. - """ - from .verses import flowverses - - whose, _, rest = named.partition("/") - waiting = [ - one.name - for one in flowverses() - if one.url and not one.fetched and (one.name == whose if rest else True) - ] - if not waiting: - return "no flow to read: a flow is a directory with an __init__.py in it" - said = " and ".join(waiting) - which = "flowverse has" if len(waiting) == 1 else "flowverses have" - return ( - f"the {said} {which} not been fetched yet -- open /flowverses and press r on it" - ) - - -def _entry(inside: dict[str, Any], wanted: str) -> Callable[..., Any] | None: - """The flow a file was asked for, out of everything in it. - - By what it was marked with and never by what it is called: a file is run to be read, and - the functions it leaves behind are its flows, whatever it imported and whatever it broke a - flow into. `@flow()` is the one the file holds under its own name. - - Args: - inside: What running the file left behind. - wanted: Which of its flows was asked for, or "" for the one it holds under its own name. - - Returns: - The entry point, or None where the file holds no such flow. - """ - from hmz._legacy_flows import Flow - - for one in inside.values(): - said = getattr(one, "__humanize_flow__", None) - if isinstance(said, Flow) and said.name == wanted: - return cast("Callable[..., Any]", one) - return None - - -def _holds(inside: dict[str, Any]) -> list[str]: - """What a file calls each of the flows it holds under a name of its own. - - Args: - inside: What running the file left behind. - - Returns: - One name apiece, in the order the file declared them. Its `run` is not among them: it - is the flow the file holds under its own name, and has no name of its own. - """ - from hmz._legacy_flows import Flow - - said = (getattr(one, "__humanize_flow__", None) for one in inside.values()) - return [one.name for one in said if isinstance(one, Flow) and one.name] - - -def set_up( - flow: str | os.PathLike[str], - setting: type[BaseModel] | None, - config: BaseModel | dict[str, Any], -) -> BaseModel: - """Reads a config back into the model the flow has just declared. - - Read back rather than taken as it comes, because a flow is loaded by running its file: - the class it declared last time is not the class it declares this time, so what was set - up against one is a stranger to the other. What survives that is the fields, which is - what a config is -- and reading them back is also what puts them through the flow's own - validators one last time, at the moment the flow is about to run. A mapping of the same - fields, which is what a YAML file of them reads as, is read back the same way. - - Args: - flow: The flow, for what a refusal says. - setting: What it says it can be set up with, or None where it said nothing. - config: What it is being set up with. - - Returns: - The same settings, as an instance of the model this loading of the flow declared. - - Raises: - NotAFlow: If the flow takes no config, or takes another one, or will not accept these - settings -- each of which is a caller to correct before anything runs. - """ - from pydantic import ValidationError - - if setting is None: - raise NotAFlow(f"{flow}: the flow takes no config, and one was given") - if not isinstance(config, dict) and type(config).__name__ != setting.__name__: - raise NotAFlow( - f"{flow}: the flow takes a {setting.__name__} to be set up with, not a " - f"{type(config).__name__}" - ) - fields = config if isinstance(config, dict) else config.model_dump() - try: - return setting.model_validate(fields) - except ValidationError as refused: - raise NotAFlow(f"{flow}: {refused}") from refused - - -def _setting(run: Entry, hinted: dict[str, object]) -> type[BaseModel] | None: - """The model a flow says it can be set up with, read off its third argument. - - Third rather than named, because that is where it is: `run(agents, task, config)` is the - entry point, and a flow which takes nothing more has two arguments and is left alone. - - Args: - run: The flow's entry point. - hinted: Its annotations, resolved. - - Returns: - The model, or None where the flow takes no third argument or annotated it with - something that is not one -- a flow is not refused for the shape of an argument - nothing has to fill. - """ - from pydantic import BaseModel - - taken = list(inspect.signature(run).parameters) - if len(taken) < _WITH_A_CONFIG: - return None - kind = hinted.get(taken[_WITH_A_CONFIG - 1]) - # `Model | None` is the annotation a flow writes, and is two arguments to a union; one - # written as the model alone is the same question with no way to answer it as unasked. - for said in (*get_args(kind), kind): - if isinstance(said, type) and issubclass(said, BaseModel): - return said - return None - - -def _kinds(declared: type, run: Entry) -> dict[str, object]: - """What a flow annotated each place of its agents with, resolved where it can be. - - Against the flow's own globals, which are where its names are: a flow loaded by running - the file is not a module anything can look up, so the class cannot resolve its own - annotations on its own. - - Args: - declared: The named tuple the flow declared its agents as. - run: Its entry point, which is what carries those globals. - - Returns: - One annotation per place, resolved if they could be resolved and as they were written - if they could not -- a name that will not resolve is still a name to read. - """ - try: - return dict( - get_type_hints( - declared, globalns=dict(run.__globals__), include_extras=True - ) - ) - except (NameError, TypeError): - return dict(getattr(declared, "__annotations__", {})) - - -def _place(name: str, kind: object) -> Place: - """One place in a flow's agents, read off what the flow annotated it with. - - Args: - name: What the flow calls it, or "" where it named none of them. - kind: The annotation, which may be an `Annotated` carrying what the flow asks of - whoever fills the place. - - Returns: - The place. - """ - moments = frozenset(_moments(kind)) - where = _where(kind) - goal = _goal(kind) - runs = _runs(kind) - needs = _needs(kind) - if get_origin(kind) is Annotated: - kind = get_args(kind)[0] - return Place( - name=name, - person=_is_person(kind), - moments=moments, - where=where, - goal=goal, - permission=runs.permission, - # A place the flow runs under a goal has one: `Goal` is the more particular of the - # two things it wrote, and a flow that asked for both ways at once is a flow the - # checker says so about rather than one that quietly does neither. - goals=True if goal else runs.goals, - web_search=runs.web_search, - insist=runs.insist, - needs=needs, - ) - - -def _where(kind: object) -> type[Remote] | Remote | Isolated | None: - """Where a flow said the agent filling a place may work. - - Args: - kind: What the flow annotated the place with. - - Returns: - What it wrote beside the type -- `Remote`, or an `Isolated` naming an image -- and None - for a place it annotated with the type alone, which is one that works here. - """ - from hmz.coganchor.agents import Isolated, Remote - - if get_origin(kind) is not Annotated: - return None - for said in get_args(kind)[1:]: - if said is Remote or isinstance(said, (Remote, Isolated)): - return said - return None - - -def _needs(kind: object) -> Needs | None: - """What a flow said filling a place takes, of the agent and of where it works. - - Args: - kind: What the flow annotated the place with. - - Returns: - The `Needs` it wrote beside the type, and None for a place it wrote none beside -- - which is one any backend may fill, working wherever the rest of the annotation allows. - """ - from hmz.coganchor.agents import Needs - - if get_origin(kind) is not Annotated: - return None - for said in get_args(kind)[1:]: - if isinstance(said, Needs): - return said - return None - - -def _goal(kind: object) -> bool: - """Whether a flow said the agent filling a place is run under its backend's goal feature. - - Args: - kind: What the flow annotated the place with. - - Returns: - True if it wrote `Goal` beside the type, and False for a place annotated with the type - alone -- which is one driven by turns like every other. - """ - from hmz.coganchor.agents import Goal - - if get_origin(kind) is not Annotated: - return False - return any(said is Goal for said in get_args(kind)[1:]) - - -def _runs(kind: object) -> AgentDefaults: - """What a flow said the agent filling a place runs at. - - Args: - kind: What the flow annotated the place with. - - Returns: - The `AgentDefaults` it wrote beside the type, and the ordinary one -- `bypass`, goals - on, the web readable -- for a place it wrote none beside, which settles nothing: those - are the loosest of each, and what an agent carries is never loosened. - """ - from hmz.coganchor.agents import AgentDefaults - - if get_origin(kind) is Annotated: - for said in get_args(kind)[1:]: - if isinstance(said, AgentDefaults): - return said - return AgentDefaults() - - -def _moments(kind: object) -> tuple[Moment, ...]: - """The moments a flow asked the agent filling a place to run. - - Args: - kind: What the flow annotated the place with. - - Returns: - Whatever moments it wrote beside the type, in the order it wrote them, and nothing at - all for a place it annotated with the type alone. - """ - from hmz.coganchor.agents import Moment - - if get_origin(kind) is not Annotated: - return () - return tuple(said for said in get_args(kind)[1:] if isinstance(said, Moment)) - - -def _is_person(kind: object) -> bool: - """Whether a place in a flow's agents is the person at the prompt. - - Args: - kind: What the flow annotated that place with, which is the class itself, or its name - where the flow put its annotations off until they are asked for. - - Returns: - True if it is a `Person`, which is a place nobody is asked to configure. The class that - answers to that interface is taken for it too: a flow written before there was one names - the driver, and the place it meant is the same place. - """ - from hmz._legacy_flows import Person - from hmz.coganchor.agents import HumanAgent - - people = (Person, HumanAgent) - if isinstance(kind, str): - # Read by the word it names rather than by what that word means, which is all there - # is to go on: the first thing inside an `Annotated[...]` is the type it is about. - said = kind.removeprefix("Annotated[").split(",")[0].strip() - return said.rpartition(".")[2] in {one.__name__ for one in people} - return any(kind is one for one in people) - - -def _about() -> dict[str, Any]: - """What is running now, for a report of something that went wrong while it was. - - Read off what is running rather than off whichever run registered last: a flow that called - another is two runs, and a crash after both have ended belongs to neither. Names and never - contents -- which flow, how long it has been going, and for each of its agents the backend - it drives, the model at the effort, the account by the name it was made under, what it may - do, where its work lands and which skills the flow mounted onto it. What the flow was told, - what any agent said and what is in any file are not here and are not reachable from what is. - - Returns: - The description, as plain values something can write out as YAML. The flow named at the - top is the one somebody started, which is the one a report is about; whatever it called - is under `running` beneath it, each saying how deep it is and what called it -- a run - is a tree, and a report of one that flattened it would say a flow ran under the wrong - one. - """ - with _TELLING: - held = [one for one in _RUNNING.values() if one.thread.is_alive()] - return { - "flow": next((one.one.flow for one in held if one.one.under is None), ""), - "running": [ - { - "flow": each_of.one.flow, - "deep": each_of.one.depth, - "under": ( - each_of.one.under.flow if each_of.one.under is not None else "" - ), - "for": round(time.monotonic() - each_of.one.since), - "agents": [ - { - "called": each.id, - "cli": each.backend, - "model": each.config.model, - "effort": each.config.effort, - "service_tier": each.config.service_tier, - "account": each.config.provider - or "as this machine is signed in", - "may": each.config.permission, - "goals": each.config.goals, - "web_search": each.config.web_search, - "works": "here" if each.config.machine is None else "elsewhere", - "skills": [loaded.name for loaded in each.loaded], - } - for each in each_of.agents - ], - } - for each_of in held - ], - } - - -# What a report of a failure carries about the run it happened in: asked for only if one is -# ever made, and never otherwise. Registered once, here, because what it answers is what is -# running at the moment of the report rather than anything one run holds. -telemetry.about("flow", _about) diff --git a/src/hmz/runtime/flowing/engine.py b/src/hmz/runtime/flowing/engine.py index 94ed7a95..d006a19b 100644 --- a/src/hmz/runtime/flowing/engine.py +++ b/src/hmz/runtime/flowing/engine.py @@ -47,6 +47,7 @@ CapabilityMissing, CostExceeded, DurationExceeded, + EnvBackendKind, FlowCancelled, FlowDefinitionError, FlowDepthExceeded, @@ -131,6 +132,9 @@ _INF = math.inf +#: The backend a `LocalEnv` role may be given an environment on. +_LOCAL = EnvBackendKind.LOCAL + #: What a role's memo answers for a grant it has not been offered yet. _UNSEEN: Any = object() @@ -368,7 +372,11 @@ async def __call__( for role in self._eroles: name = role.name given = envs.get(name) - if type(given) is EnvView and not role.resources: + if ( + type(given) is EnvView + and not role.resources + and (not role.auto or given._driver.backend == _LOCAL) + ): grant = given._grant if grant is role.grant or role._seen.get(grant, _UNSEEN) is None: env_views[name] = EnvView( @@ -578,6 +586,12 @@ def _env(role: EnvRole, given: object, node: Call, flow: FlowImpl) -> EnvView: f"{flow.ref}: {role.name!r} was given {given!r}, which is not an environment " "the run handed out" ) + driver = given._driver + if role.auto and driver.backend != _LOCAL: + raise CapabilityMissing( + f"{flow.ref}: {role.name!r} is a LocalEnv, and the environment given is " + f"{driver.backend}@{driver.provider}{driver.workdir}, which is not this machine" + ) return _narrowed( role, given._driver, diff --git a/src/hmz/runtime/flowing/finding.py b/src/hmz/runtime/flowing/finding.py index 4b051edf..5e830469 100644 --- a/src/hmz/runtime/flowing/finding.py +++ b/src/hmz/runtime/flowing/finding.py @@ -1,4 +1,4 @@ -"""Where a flow is, what it is called, and what running its file leaves behind. +"""Where a flow is, what it is called, and the flow a name comes to. A flow is named rather than pathed: `hmz exec -f ralph_loop` is a name, and a path is what is left for a flow that is nowhere any of them are kept. A name is looked for in the places flows @@ -11,199 +11,61 @@ of your own may stand in for one of humanize's by taking its name, and `local/chat` is the spelling that says which one it is. -Reading a flow means running it. A flow is a Python file, what it holds is whatever marking a -function with :func:`~hmz._legacy_flows.flow` left behind, and the only way to find that out is -to run -the file -- with its own directory importable while it does and only while, and forgotten again -afterwards, so that the module beside one flow is never answered with the module beside -another. Run afresh every time, too: a flow rewritten between two runs of it -- by hand, or by -an agent it is itself driving -- is the flow that runs next. - -None of this is a thing a flow names. A flow says what it is with the mark, and humanize does -the finding: :mod:`hmz._legacy_flows` is the whole of what a flow imports, and everything that -reads a -flow is here, written against it. +Reading a flow means importing it: a flow is a directory whose `__init__.py` defines flows with +:func:`hmz.flows.flow`, and the only way to find out which is to import it. That is +:mod:`loading`'s to do, once per module and afresh when its files change, so a flow rewritten +between two runs of it -- by hand, or by an agent it is itself driving -- is the flow that runs +next. """ from __future__ import annotations -import contextlib import os -import runpy -import sys from pathlib import Path -from typing import TYPE_CHECKING, Any, NamedTuple - -from hmz import _legacy_flows -from hmz._legacy_flows import ( - _SAID, # pyright: ignore[reportPrivateUsage] - Flow, - _first, # pyright: ignore[reportPrivateUsage] -) +from typing import TYPE_CHECKING, NamedTuple from .verses import LOCAL, MINE, OFFICIAL, flowverses, holds, nearest if TYPE_CHECKING: + from .engine import FlowImpl from .verses import Flowverse __all__ = [ "BUILTIN_AT", "ENTRY", - "PROPHECY", "Offer", "about", "at", + "builtin", "entry", "find", - "foretold", "fork", "found", - "held", "inside", - "loaded", "offered", "offers", - "reading", + "resolved", "within", ] -#: Where the flows humanize ships in the package are: a directory of them inside -#: :mod:`hmz._legacy_flows`, which is where a flow lives, rather than beside this file, which is how -#: one is found. They are the whole of what is there, so there is no `flows/` in it to tell -#: them from the rest. Offered under `official` along with the repository of the rest of -#: humanize's flows: which of the two places one of them is kept in is humanize's business -#: rather than whoever is running it. -BUILTIN_AT = Path(_legacy_flows.__file__).parent / "builtin" +#: Where the flows humanize ships in the package are: `hmz/flows/builtin`, beside the flow API +#: they are written against, rather than beside this file, which is how one is found. They are +#: the whole of what is there, so there is no `flows/` in it to tell them from the rest. Offered +#: under `official` along with the repository of the rest of humanize's flows: which of the two +#: places one of them is kept in is humanize's business rather than whoever is running it. +#: Worked out from where this file is rather than by importing the flow API, which a listing +#: of places has no need to pay for. +BUILTIN_AT = Path(__file__).resolve().parents[2] / "flows" / "builtin" #: What a flow's directory holds the flow itself in. The rest of the directory is what it #: imports and the `skills/` it brings, so the entry point is named rather than guessed. ENTRY = "__init__.py" -#: And what an atlas's directory may hold the prophecy it was already compiled to in. A -#: flowverse that ships one ships the graph its flow was checked into, and that graph is -#: what runs: the compiling is where an atlas is refused, and a repository which has been -#: through it once has an answer worth carrying rather than working out again. -PROPHECY = "prophecy.pkl" - -#: What a flow's own name is separated from the one inside it by. A flow that holds one flow -#: is named by itself; one that holds three names each of them after it. +#: What a flow's own name is separated from the one inside it by. A module that holds one +#: visible flow is named by itself; one that holds three names the others after it. _INSIDE = ":" -def loaded(where_: str | os.PathLike[str]) -> dict[str, Any]: - """Runs a flow's entry point and answers with what it left behind. - - With its own directory importable while it runs, and only while: a flow is a directory of - what it needs, and one that reaches for the module next to it is reaching for something - that came with it. The directory the flows are in is importable too, for what a flowverse - keeps beside them for all of them. Put back afterwards, since what a flow imports is not - something the rest of this process should be able to. - - Run each time rather than cached: a flow rewritten while a run is going is the flow that - runs next, which is what lets a flow -- or an agent driving one -- rewrite it and go on. - - Args: - where_: The flow: its directory, or the Python file to run outright. - - Returns: - Everything running it defined, by name. - """ - where_ = os.path.join(where_, ENTRY) if os.path.isdir(where_) else where_ - beside = os.path.dirname(os.path.abspath(where_)) - among = os.path.dirname(beside) - sys.path[:0] = [beside, among] - try: - return runpy.run_path(str(where_)) - finally: - for one in (beside, among): - with contextlib.suppress(ValueError): - sys.path.remove(one) - _forgotten(beside, among) - - -def _forgotten(*under: str) -> None: - """Forgets what was imported from beside a flow, so nothing of it outlives the run. - - A flow imports the module next to it by its plain name -- `import prompts` -- and every - flow may have one. Left in `sys.modules`, the first flow loaded in a process owns that name - for the life of it: the next flow's `import prompts` is answered with the last one's, and a - menu drawing the list of flows is enough to settle who won. Taken out, each run of a flow - reads what is beside that flow -- which is also what makes a flow edited between two runs - of it run as it is now, module beside it and all. - - What is dropped is only what was loaded out of these directories, found by the file each - module says it came from. Nothing of humanize's own is: a flow kept inside humanize's own - tree would otherwise unload the package that is running it. - - Args: - under: The directories, as absolute paths. - """ - roots = tuple(one + os.sep for one in under) - for name, module in list(sys.modules.items()): - if name.startswith("hmz"): - continue - at = getattr(module, "__file__", None) - if at and os.path.abspath(at).startswith(roots): - del sys.modules[name] - - -def held(where_: str | os.PathLike[str]) -> list[Flow]: - """Every flow one file holds: its own first, and the rest as it declares them. - - Args: - where_: The flow -- its directory, or the file to read outright. It is run to be read, - so whatever it does as it is imported happens here. - - Returns: - One per function it marked with :func:`flow`, the one it marked with no name first -- - which is the flow the file holds under its own name. Nothing at all for a file that - marks none, or cannot be read: this is asked while a list is being drawn, and a file - that will not import is one line of that list rather than the end of it. - """ - try: - inside = loaded(where_) - except Exception: # noqa: BLE001 -- a file that will not run holds no flows to list - return [] - return _flows_of(inside) - - -def _flows_of(inside: dict[str, Any]) -> list[Flow]: - """Every flow in what running one file left behind. - - Args: - inside: What the file defined, by name. - - Returns: - One per function the file marked with :func:`flow`, in the order it declared them -- - which for three phases of one thing is their order -- and the one it marked with no name - first, since that is the one the file is named after and a list that put it third would - read as the third thing in the file. Nothing at all for a file that marks nothing, which - a directory of flows may well have in it: something the flows beside it import, or the - file that sets their tests up. A name declared twice is the first of them: a file that - holds two flows of one name is a file to correct, and picking one of them at random is - not the way to say so. - """ - said: list[Flow] = [] - for one in inside.values(): - marked = getattr(one, _SAID, None) - if not isinstance(marked, Flow) or any( - marked.name == already.name for already in said - ): - continue - # The file's own docstring where the flow it holds says nothing: a file that is one - # flow is documented as that flow, and its first line is what it does. - if not marked.name and not marked.about: - marked = Flow( - name="", - about=_first(inside.get("__doc__")), - skills=marked.skills, - resumable=marked.resumable, - selectable=marked.selectable, - ) - said.append(marked) - return [one for one in said if not one.name] + [one for one in said if one.name] - - class Offer(NamedTuple): """One flow there is to run, as whatever is offering them lists it. @@ -327,19 +189,21 @@ def offers(one: Flowverse) -> list[Offer]: that moves between the two goes on answering to the name it always had. Yours are named the same way as anybody else's -- `local/scheduler`, `user/scheduler` -- so that a flow of yours sharing a name with one of humanize's is listed beside it under a name of its own - rather than instead of it. A flow that holds several names each of them, - `:` apiece, and a directory that holds none is not among them -- a directory - of flows has directories beside them that are not one. + rather than instead of it. The flow a bare ref names -- the one named after its + directory, else the only visible one -- is listed by the directory's name, and every + other visible flow of the module as `:`; hidden flows are not listed, and a + directory that holds none is not among them -- a directory of flows has directories + beside them that are not one. Just the ones in the package for `official` before it has been fetched, and nothing at all for any other flowverse that has not been, which is not the same answer as one that holds nothing, and is why :class:`Flowverse` says which it is. Note: - Reading a flow means running it, so the entry point of every flow in the directories the - flowverse holds its flows in is run to find out what it holds -- and nothing outside - them, which is what those directories are for. Whoever added it is trusting that - repository with this machine; this is where that trust is spent. + Reading a flow means importing it, so the entry point of every flow in the directories + the flowverse holds its flows in is imported to find out what it holds -- and nothing + outside them, which is what those directories are for. Whoever added it is trusting + that repository with this machine; this is where that trust is spent. """ from .verses import flows @@ -352,29 +216,51 @@ def offers(one: Flowverse) -> list[Offer]: def _named(at: Path, called: str) -> list[tuple[str, str]]: - """What each flow in one file is called, given what the file itself is called. + """What each flow in one module is called, given what the module itself is called. Args: - at: The file. - called: What the file is called where it was found. + at: The module's entry point. + called: What the module is called where it was found. Returns: - One `(name, what it says about itself)` pair per flow: the file's own name for the flow - it holds under it, and `:` for each of the rest. Nothing at all for a - file that holds no flow -- a directory of flows has files beside them that are not one -- - but just the file's name for one that could not be read: a file that will not import is - still a flow somebody named, and saying so where they pick it is better than leaving it - off the list. + One `(name, what it says about itself)` pair per visible flow: the module's own name for + the one a bare ref names, and `:` for each of the rest. Nothing at all + for a module that defines no flow -- a directory of flows has files beside them that are + not one -- but just the module's name for one that could not be imported: a file that + will not import is still a flow somebody named, and saying so where they pick it is + better than leaving it off the list. """ + from hmz.flows import FlowException + + from .loading import module_of, pick + try: - inside = loaded(at) - except Exception: # noqa: BLE001 -- named as a flow, and not readable to be sure it is + module = module_of(at, None) + flows = module.flows() + except (FlowException, OSError): return [(called, "")] - return [ - (called if not one.name else f"{called}{_INSIDE}{one.name}", one.about) - for one in _flows_of(inside) - if one.selectable - ] + visible = [one for one in flows.values() if not one.hidden] + try: + bare = pick(module, "", called) + except FlowException: + bare = None + said: list[tuple[str, str]] = [] + if bare is not None and not bare.hidden: + # The module's own docstring where the flow it is named for says nothing: a module + # that is one flow is documented as that flow, and its first line is what it does. + said.append((called, bare.description or _first(module.module.__doc__))) + said.extend( + (f"{called}{_INSIDE}{one.name}", one.description or "") + for one in sorted(visible, key=lambda one: one.name) + if one is not bare + ) + return said + + +def _first(doc: str | None) -> str: + """The first line of a docstring, or "" for none.""" + said = (doc or "").strip().splitlines() + return said[0].strip() if said else "" def about(named_: str) -> str: @@ -386,11 +272,110 @@ def about(named_: str) -> str: Returns: The line, or "" for a flow that says nothing or cannot be read. """ - at, inside = _split(named_) - for one in held(find(at)): - if one.name == inside: - return one.about - return "" + from hmz.flows import FlowException + + try: + flow = resolved(named_) + except (FlowException, OSError): + return "" + if flow.description or inside(named_) or flow.home is None: + return flow.description or "" + # The module's own docstring where the flow a bare name means says nothing, as the list + # of flows says it: a module that is one flow is documented as that flow. + return _first(flow.home.module.__doc__) + + +def resolved(named_: str) -> FlowImpl: + """The flow a name comes to, loaded, as a way in runs it. + + Anything :func:`hmz.flows.load` takes where no flow is asking: a name nearest first, + `/`, either with `:`, a path, or a `git+#` ref, which + is fetched here, on this thread. A flow humanize ships is handed its harness's every + capability, with :func:`~hmz.runtime.flowing.engine.full_view`: `chat` talks to whichever + agent it is given, and so declares nothing of any, and it is the flows in the package + rather than a name that says which flow that is. + + Args: + named_: What the flow is called. + + Returns: + The flow. + + Raises: + FlowRefError: If `named_` is no ref, or is relative to a flow when none is asking. + FlowNotFound: If nothing answers to it. + FlowDefinitionError: If what it names is written wrong or will not import. + FlowLoadConflict: If loading it would replace a module a run going now uses. + """ + from hmz.flows import FlowNotFound + + from .engine import FlowImpl, full_view + from .loading import load + + try: + found_ = load(named_, {}) + except FlowNotFound as missing: + # Only for a name nothing answers to: one that found its module and then no flow in + # it is a flow to correct, whatever has been fetched. + if "#" in named_ or os.path.isfile(find(named_)): + raise + waiting = _unfetched(named_) + if not waiting: + raise + raise FlowNotFound(f"{named_}: {waiting}") from missing + if isinstance(found_, FlowImpl): + flow = found_ + else: + # Fetched on this thread: a way in asks before anything runs, and has no loop yet. + _ = found_.name + fetched = found_.flow + assert fetched is not None # noqa: S101 -- asking its name fetched it + flow = fetched + if builtin(flow): + full_view(flow) + return flow + + +def _unfetched(named_: str) -> str: + """Why a flow that was named is not there, where a flowverse not fetched yet is why. + + A flowverse is offered before it is fetched -- `official` is there from the start -- so + "no such flow" would be the answer to a name that is right, given by the one thing that + knows it has not been downloaded. A name that said which place it came from is a question + about that place alone; a bare one is looked for in every one of them. + + Args: + named_: What was asked for, as it was written. + + Returns: + The reason, or "" where every flowverse it could have come from has been fetched. + """ + whose, _, rest = _split(named_)[0].partition("/") + waiting = [ + one.name + for one in flowverses() + if one.url and not one.fetched and (one.name == whose if rest else True) + ] + if not waiting: + return "" + which = "flowverse has" if len(waiting) == 1 else "flowverses have" + return ( + f"the {' and '.join(waiting)} {which} not been fetched yet -- open /flowverses " + "and press r on it" + ) + + +def builtin(flow: FlowImpl) -> bool: + """Whether a flow is one humanize ships in the package. + + Args: + flow: The flow. + + Returns: + Whether it was defined under :data:`BUILTIN_AT`. + """ + made = Path(os.path.realpath(flow.fn.__code__.co_filename)) + return made.is_relative_to(BUILTIN_AT) def _split(named_: str) -> tuple[str, str]: @@ -453,54 +438,6 @@ def find(named_: str) -> str: return at_ -def reading(named_: str) -> str: - """What to point a reading of one flow at, which is not always what runs it. - - A flow is a directory or a single file, and the two readings of one -- the checking and - the compiling -- take the whole of it either way: the directory where there is one, so - that what the entry point imports beside it is read too, and the file where there is - not. :func:`find` answers with the entry point instead, that being what is run. - - Args: - named_: A flow's name, as :func:`find` takes it. - - Returns: - The path to read: the flow's own directory, or the file a single-file flow is. A name - nothing answers to comes back as :func:`find` left it, so whatever asked hears about - it where it looks rather than here. - """ - found_ = find(named_) - if os.path.isfile(found_) and os.path.basename(found_) == ENTRY: - return os.path.dirname(found_) - return found_ - - -def foretold(named_: str) -> str: - """Where the prophecy one flow ships is, for a flow that ships one. - - An atlas is compiled before it runs, and a flowverse may ship what compiling it came - to: `prophecy.pkl`, beside the entry point, holding the graph the atlas was read into. - Where there is one it is what runs -- the compiling having already happened, in the - repository the flow came from, over the source that repository holds. - - What is beside it still matters. A prophecy names the functions its nodes are, and - those are in the flow's own Python: a directory holding a prophecy and no entry point - is not a flow, the same way a directory holding neither is not one. - - Args: - named_: A flow's name, as :func:`find` takes it. - - Returns: - The path to it, and "" for a flow that ships none -- which is every flow that is not - an atlas, and most atlases. - """ - from .prophecy import shipped - - beside = at(named_) - held = shipped(beside) if beside else None - return "" if held is None else str(held.at) - - def at(named_: str) -> str: """The flow's own directory, which is where what it brings with it lives. diff --git a/src/hmz/runtime/flowing/prophecy.py b/src/hmz/runtime/flowing/prophecy.py deleted file mode 100644 index 292e0dbd..00000000 --- a/src/hmz/runtime/flowing/prophecy.py +++ /dev/null @@ -1,433 +0,0 @@ -"""What an atlas compiles to: the prophecy a run of one walks, and the file it ships in. - -An atlas is written with the marks in :mod:`hmz._legacy_flows.atlas` and read by -:mod:`hmz.runtime.flowing.prophesying`. This is what that reading answers with -- the nodes, -the edges, the shapes that flow along them -- and what :mod:`hmz.runtime.flowing.stepping` -walks a run over. None of it is a thing an atlas author writes, which is why it is here and -not beside the marks: a flow names what it declares, and never what its declaration was -turned into. - -A prophecy is canonical: the same atlas written twice the same way compiles to the same text, -byte for byte, and :func:`digest` over that text is what a run picked up again checks itself -against. An atlas rewritten between two runs is a different prophecy, and a run that carried -on into it would be a run resuming into somewhere it had never been. - -A flowverse may ship one beside the atlas it compiled, which is `prophecy.pkl`: the compiling -is where an atlas is refused, so a repository that has been through it has an answer worth -carrying. It is read back here and nowhere else -- and read back to this module's own tuples -and nothing else, since the file is opened by a reading whose whole promise is that it -executes nothing. -""" - -from __future__ import annotations - -from pathlib import Path -from typing import TYPE_CHECKING, Any, NamedTuple - -if TYPE_CHECKING: - import os - - from hmz._legacy_flows.atlas import Kind - -__all__ = [ - "AGENTS", - "CONFIG", - "INPUT", - "Edge", - "Field", - "Node", - "Prophecy", - "Reads", - "Shape", - "Shipped", - "When", - "canonical", - "digest", - "kept", - "shipped", - "told", -] - -#: What a node reads when it is handed one of the run's agents rather than a value: the -#: agents are what the run was started with rather than anything a node answered, so they -#: are named where a node id would be. Not an identifier, so nothing an atlas can write -#: collides with it. -AGENTS = "@agents" - -#: And what it reads when it is handed the flow's own input -- the task a command line gave, -#: or the shape a supernode was called with. -INPUT = "@input" - -#: And what it reads when it is handed what the run was set up with, for an atlas that says -#: it takes a config. -CONFIG = "@config" - - -class Field(NamedTuple): - """One field of one shape, as the compiling read it off the model that declares it. - - Attributes: - name: What the field is called. - shape: The shape it holds, by name. - required: Whether the model refuses to be built without it, which is what an edge is - held to: what flows in has to cover what the far end cannot do without. - """ - - name: str - shape: str - required: bool - - -class Shape(NamedTuple): - """One thing that may flow along an edge, read off the atlas's own files. - - Attributes: - name: The model's name, or the plain kind -- `str`, `int`, `float`, `bool`. - fields: One per field the model declares, in the order it declares them, and nothing at - all for a plain kind, which has none. - """ - - name: str - fields: tuple[Field, ...] = () - - -class Reads(NamedTuple): - """Where one of a node's arguments comes from. - - A name rather than the node that answered it, because a body may bind a name twice -- - which is what a loop is, the second binding being the one the next round reads. So a run - keeps what each name holds now, and a node says which of them it wants. - - Attributes: - reads: The name it reads: one the body bound, or :data:`AGENTS` for the run's agents, - :data:`INPUT` for what the flow itself was called with, and :data:`CONFIG` for what - it was set up with. - field: The field read off it, and "" for the whole of it. - """ - - reads: str - field: str = "" - - -class When(NamedTuple): - """What has to hold for one edge to be the way out that is taken. - - Attributes: - reads: The name the branch reads, which is one a node bound. - field: The field read off it, and "" for the whole of it. - truth: Whether this is the way out taken when that reads as true or as false. - """ - - reads: str - field: str - truth: bool - - -class Node(NamedTuple): - """One node of a prophecy: one call site of the body it was compiled from. - - A node is a call site rather than a function, since a body that calls one function twice - is a prophecy with two nodes in it -- each with its own answer, its own place in the run, - and its own line in what a run picked up again has already done. - - Attributes: - at: The node id: what it calls, and `:2`, `:3` after it where the body calls that same - thing more than once. Read off the body's shape rather than off a line number, so - that a file reformatted compiles to the prophecy it already was. - kind: Which of the three it is. - calls: The function it runs, by the name the atlas's own files declare it under -- or, - for a supernode from another file, the flow by the name `-f` takes. - takes: Where each of its arguments comes from, in the order it takes them. - binds: The name its answer is bound to, and "" for a node whose answer nothing takes. - gives: The shape it answers with, and "" for a node that answers with nothing. - rerun: Whether a run picked up again runs it again where the last run stopped inside - it, or steps past it. - under: For a supernode, the prophecy it is, by the name that prophecy is called. "" for - every other node. - """ - - at: str - kind: Kind - calls: str - takes: tuple[Reads, ...] = () - binds: str = "" - gives: str = "" - rerun: bool = True - under: str = "" - - -class Edge(NamedTuple): - """One way from one node to the next. - - Attributes: - out_of: The node it leaves, and "" for the way into the prophecy. - into: The node it arrives at, and "" for the way out of it, which is where the run - ends. - when: What has to hold for this to be the way taken, and None for a node's only one. - answers: For a way out of the prophecy, the name the run answers with -- which is what - the `return` named, and not whatever the last node happened to say. "" everywhere - else, and for an atlas that answers with nothing. - """ - - out_of: str - into: str - when: When | None = None - answers: str = "" - - -class Prophecy(NamedTuple): - """One atlas, compiled: the whole of what a run of it will do. - - Attributes: - name: The flow, as it was asked for -- which for a supernode of another file is the - name `-f` takes, and for one beside it is that file's own name for it. - takes: The shape the flow is called with, which is `str` for one a command line runs - and a model for one that is only ever a supernode. - gives: The shape it answers with, and "" for one that answers with nothing. - config: The shape it is set up with, and "" for one that takes no setting up -- which - every supernode is, what is set up being the run rather than a node of it. - agents: What the atlas calls each of the agents it drives, in the order it takes them. - nodes: Every node, by node id. - edges: Every way from one node to another, the way in and the way out included. - shapes: Every shape anything in it carries, the ones its supernodes carry included. - prophecies: One per supernode, which is the sub-atlas that node is. - """ - - name: str - takes: str - gives: str - config: str - agents: tuple[str, ...] - nodes: tuple[Node, ...] - edges: tuple[Edge, ...] - shapes: tuple[Shape, ...] - prophecies: tuple[Prophecy, ...] = () - - def node(self, at: str) -> Node | None: - """The node of that id, or None where the prophecy holds none. - - Args: - at: The node id. - - Returns: - The node. - """ - return next((one for one in self.nodes if one.at == at), None) - - def out_of(self, at: str) -> tuple[Edge, ...]: - """Every way out of one node, in the order they are to be tried. - - Args: - at: The node id, or "" for the way into the prophecy. - - Returns: - The edges, the guarded ones first: a node with a branch and a way out that is - taken otherwise is read as the branch it is rather than as a coin toss. - """ - found = [one for one in self.edges if one.out_of == at] - return tuple(sorted(found, key=lambda one: one.when is None)) - - def under(self, named: str) -> Prophecy | None: - """The sub-prophecy of that name, or None where this prophecy holds none. - - Args: - named: What the supernode said it was. - - Returns: - The prophecy. - """ - return next((one for one in self.prophecies if one.name == named), None) - - -def canonical(prophecy: Prophecy) -> str: - """One prophecy as the text two readings of the same atlas both answer with. - - Canonical means what it says: everything ordered by what it is rather than by where it - was written, so a body reformatted, a comment added or two nodes swapped where nothing - depends on the order compile to the same bytes. That is what makes :func:`digest` worth - keeping -- a run picked up again asks whether the atlas is still the atlas it was, and an - answer that changed when somebody reflowed a docstring would be no answer. - - Args: - prophecy: The compiled atlas. - - Returns: - JSON, keys sorted, one line: what a script diffs and what a person reads. - """ - import json - - return json.dumps(_written(prophecy), sort_keys=True, ensure_ascii=False) - - -def _written(prophecy: Prophecy) -> dict[str, Any]: - """One prophecy as the plain objects :func:`canonical` writes out. - - Read off the tuples themselves rather than field by field: everything here is a - NamedTuple, so a field added to one later is a field the canonical text carries and the - digest sees -- where a hand-written list of them would drop it without saying so, and - two prophecies that differ would hash the same. - - Args: - prophecy: The compiled atlas. - - Returns: - Its nodes by id, its edges in order, its shapes by name and the prophecies under it by - name -- each sorted, since the order a body happens to be written in is not part of - what the atlas is. - """ - return prophecy._asdict() | { - "nodes": [one._asdict() for one in sorted(prophecy.nodes)], - "edges": [one._asdict() for one in sorted(prophecy.edges, key=_ordered)], - "shapes": [one._asdict() for one in sorted(prophecy.shapes)], - "prophecies": [ - _written(one) - for one in sorted(prophecy.prophecies, key=lambda one: one.name) - ], - } - - -def _ordered(edge: Edge) -> tuple[str, str, tuple[str, str, bool], str]: - """One edge as something two of them can be sorted by, an absent guard and all.""" - return (edge.out_of, edge.into, edge.when or ("", "", False), edge.answers) - - -#: What a shipped prophecy is written with. Fixed rather than highest, so that the same -#: prophecy written by two installations is the same bytes -- a flowverse ships one, and a -#: file whose contents moved under a Python upgrade is a file every checkout re-writes. -_PROTOCOL = 5 - - -def kept(prophecy: Prophecy) -> bytes: - """One prophecy as the bytes a flowverse ships beside the atlas it compiled. - - Args: - prophecy: The compiled atlas. - - Returns: - What goes in `prophecy.pkl`. - """ - import pickle - - return pickle.dumps(prophecy, protocol=_PROTOCOL) - - -#: The only classes a shipped prophecy is allowed to name. A pickle says which class to -#: build as it goes, and the reader that took it at its word would run whatever the file -#: asked for -- which the static reading of a flow, whose whole promise is that it executes -#: nothing, must not do for a file it found in a directory it was pointed at. -_SHAPES = frozenset({"Edge", "Field", "Node", "Prophecy", "Reads", "Shape", "When"}) - - -def told(said: bytes) -> Prophecy | None: - """One shipped prophecy read back, or None where those bytes are not one. - - Note: - Nothing but a prophecy is built. A pickle names the class to build at every step, so - one read as it comes runs whatever the file names -- and this file is read by the - static reading of a flow, which is pointed at code nobody has read and promises to - execute none of it. So the classes are held to this module's own tuples, and bytes - naming anything else are bytes that are not a prophecy. - - Args: - said: The bytes. - - Returns: - The prophecy, or None for bytes that are not one -- truncated, written by something - else, written by a humanize whose prophecies had another shape, or naming a class no - prophecy is made of. - """ - import io - import pickle - import sys as running - - class _Only(pickle.Unpickler): - """An unpickler that builds this module's own tuples and refuses everything else.""" - - def find_class(self, module: str, name: str) -> Any: - """Refuses every class a prophecy is not made of. - - Args: - module: The module the bytes name. - name: The class in it they name. - - Returns: - The class, for the tuples a prophecy is made of. - - Raises: - UnpicklingError: For anything else, which is what makes reading this safe. - """ - if module == __name__ and name in _SHAPES: - return getattr(running.modules[__name__], name) - raise pickle.UnpicklingError(f"a prophecy is not made of {module}.{name}") - - try: - held = _Only(io.BytesIO(said)).load() - except Exception: # noqa: BLE001 -- anything a pickle raises is a file that is not one - return None - if not isinstance(held, Prophecy): - return None - try: - canonical(held) - except (AttributeError, TypeError, ValueError): - # A named tuple of the right class holding the wrong things: written by a humanize - # whose nodes had another shape, which is a prophecy to compile again rather than - # one to walk. - return None - return held - - -class Shipped(NamedTuple): - """What one flow's own directory ships beside its entry point. - - Attributes: - at: The file it is in, which is `prophecy.pkl` beside the flow. - prophecy: What it says, and None for bytes that are not a prophecy at all -- which is - a file to compile again rather than a graph to guess at. Every reader of a shipped - prophecy decides that for itself: one refuses the run, one says so as a finding. - """ - - at: Path - prophecy: Prophecy | None - - -def shipped(under: str | os.PathLike[str]) -> Shipped | None: - """The prophecy one flow's own directory ships, where it ships one. - - The one place `prophecy.pkl` is opened. Where it is, whether it is there, and what it - takes to read it back are one rule rather than one per reader -- and what to do about a - file that will not read back is each reader's own, since a run refuses and a checking - says so. - - Args: - under: The flow's own directory. A flow that is a single file has none, and passing - the file is answered the same way as passing a directory with nothing in it. - - Returns: - Where it is and what it says, or None where the flow ships nothing. - """ - from .finding import PROPHECY - - at = Path(under) / PROPHECY - if not at.is_file(): - return None - return Shipped(at, told(at.read_bytes())) - - -def digest(prophecy: Prophecy) -> str: - """What one compiled atlas is, in sixteen characters. - - What it is for is a run picked up again: what a run has already done is written down - against the prophecy it was doing it in, and an atlas rewritten between two runs of it is a - different prophecy whose nodes happen to share their names. Carrying on into it would be a - run resuming into somewhere it has never been, so the digest is checked and a run whose - prophecy has moved starts from the top. - - Args: - prophecy: The compiled atlas. - - Returns: - The first sixteen hex characters of the SHA-256 of :func:`canonical`. - """ - import hashlib - - return hashlib.sha256(canonical(prophecy).encode()).hexdigest()[:16] diff --git a/src/hmz/runtime/flowing/prophesying.py b/src/hmz/runtime/flowing/prophesying.py deleted file mode 100644 index 58338d5e..00000000 --- a/src/hmz/runtime/flowing/prophesying.py +++ /dev/null @@ -1,1750 +0,0 @@ -"""Compiling an atlas: the reading that turns a body into the prophecy a run walks. - -An ordinary flow is read by running it, and the one thing nothing can ask it is what it is -about to do. An atlas answers that question before anything runs: its body is a declaration -in a narrower Python, and this is the reading that holds it to that Python and compiles what -it declared into an :class:`~hmz.runtime.flowing.prophecy.Prophecy`. - -Pure `ast`, like :mod:`hmz.runtime.flowing.checking`, and for the same reason: the atlas most worth -compiling is one nobody has read yet -- generated, fetched, forked -- and a compiler that ran -what it was compiling would be the attack it exists to catch. So an atlas is read, and -compiled, and only then loaded to be run. - -Every rule here is an error, and every one of them is decidable. That is the bargain an atlas -makes. The reading of an ordinary flow proves absences one function at a time and warns where -it cannot be sure; an atlas is written in the subset where there is nothing to be unsure -about. What flows along an edge either fits what the far end takes or it does not; a branch -either hangs off a logic node or it does not. That reading's warnings still come back over -the node bodies, and still do not block: a node body is ordinary Python, and is read as it. - -The subset, in one place. A body holds only these: - -- ``x = call(a, b)`` and ``call(a, b)`` -- one node apiece, whose arguments are names the - body has bound, fields read off them, or one of the flow's own three: the agents, what it - was called with, what it was set up with. -- ``if x:`` and ``if not x.field:``, with an ``else`` or without, which is a node's several - ways out. -- ``while x:`` and ``while not x.field:``, which is that with an edge back to the node that - answered the name being read. -- ``return`` and ``return x``, which is where the run ends. -- ``pass``, and the docstring. - -And nothing else. Arithmetic, comprehensions, ``try``, ``with``, ``import``, a call written -inside another call: each of them is a thing a node does, and a node is where each of them -goes. An atlas is the shape of the work rather than the work. -""" - -from __future__ import annotations - -import ast -from pathlib import Path -from typing import TYPE_CHECKING, NamedTuple - -# The reading beside this one, whose parsing, whose rules and whose small readings of a tree -# this shares: an atlas is a flow, and the whole of what makes it one is read there. Its -# public surface is what a flow-checker is asked for, and these are two readings of one -# package sharing what one of them wrote down -- not a second copy of it, kept here to drift. -from .checking import ( - Finding, - _annotated, # pyright: ignore[reportPrivateUsage] - _elements, # pyright: ignore[reportPrivateUsage] - _Mark, # pyright: ignore[reportPrivateUsage] - _Node, # pyright: ignore[reportPrivateUsage] - _parsed, # pyright: ignore[reportPrivateUsage] - _Read, # pyright: ignore[reportPrivateUsage] - _root, # pyright: ignore[reportPrivateUsage] - _rules, # pyright: ignore[reportPrivateUsage] - _tip, # pyright: ignore[reportPrivateUsage] - _unquoted, # pyright: ignore[reportPrivateUsage] - _Whole, # pyright: ignore[reportPrivateUsage] - _whole, # pyright: ignore[reportPrivateUsage] -) -from .prophecy import ( - AGENTS, - CONFIG, - INPUT, - Edge, - Field, - Node, - Prophecy, - Reads, - Shape, - When, - digest, -) - -if TYPE_CHECKING: - import os - - from hmz._legacy_flows.atlas import Kind - -__all__ = ["Prophesied", "is_atlas", "named_as", "prophesied"] - -#: The shapes an atlas may carry that are not models: the plain kinds a node may take and -#: answer with. Anything else has fields, and a thing with fields is a model -- so that what -#: flows along an edge is something both ends can be held to. -PLAIN = ("str", "int", "float", "bool") - -#: What a node that answers with nothing writes where a shape would go. -NOTHING = "None" - -#: What one node's parameter says where a shape would be, for the two places an atlas hands -#: over what the run was started with rather than anything a node answered: the agent a mind -#: drives, and the whole tuple of them a supernode is handed. -ONE_AGENT = "@agent" -THE_AGENTS = "@agents" - -#: How many things an atlas's entry point takes: the agents and what it is called with, and -#: for one that says it can be set up, the config after them. -_TAKES = 2 -_AND_A_CONFIG = 3 - - -class Prophesied(NamedTuple): - """What compiling one atlas came to. - - Attributes: - findings: One per thing the reading found, in file order. Every error among them is a - reason the atlas did not compile. - prophecy: The compiled atlas, or None where anything was an error: a graph built out of - a body the reading refused would be a graph of something nobody wrote. - """ - - findings: tuple[Finding, ...] - prophecy: Prophecy | None - - -def prophesied( - flow: str | os.PathLike[str], - *, - name: str = "", - whole: _Whole | None = None, - through: tuple[tuple[str, str], ...] = (), -) -> Prophesied: - """Reads an atlas without running it, and compiles what its body declared. - - Args: - flow: The atlas: its directory, or the Python file a single-file one is. - name: Which of the atlases the file holds, and "" for the one it holds under its own - name -- the half after the colon in `official/review:pass`. - whole: The files already parsed, for a caller that has read them, which is - :func:`hmz.runtime.flowing.checking.checked` handing on the reading it has already done. - through: The atlases this one is being compiled inside -- where each is and what it - was named -- so that a supernode reaching back into one of them is refused rather - than followed forever. - - Returns: - The findings and, where none of them is an error, the prophecy. - """ - whole = _whole(flow) if whole is None else whole - if not whole.compiled or whole.entered is None: - return Prophesied( - ( - _said( - "not-an-atlas", - whole.entry, - 0, - "nothing in it is marked @atlas -- an atlas is a flow whose body is " - "compiled, which is how a file says which of its flows is one", - ), - ), - None, - ) - rules = _rules(whole) - mark = next( - (one for one in whole.entered.marks if one.atlas and one.name == name), None - ) - if mark is None and any(one.name == name for one in whole.entered.marks): - # A flow of this file's, and an ordinary one: a file holds several flows and the - # reading each gets is its own. So the one asked for gets the reading that reads a - # body as a program, which is the reading it would have got had the file beside it - # held no atlas at all. - return Prophesied(rules, None) - found = list(rules) - for read in whole.read: - found.extend(_dynamic(read)) - if mark is None: - held = sorted(one.name for one in whole.entered.marks if one.atlas) - found.append( - _said( - "not-an-atlas", - whole.entry, - 0, - f"nothing in it is an atlas called {name!r} -- it holds " - f"{', '.join(repr(one) for one in held)}", - ) - ) - return Prophesied(tuple(found), None) - prophecy, said = _compiled( - whole, _gathered(whole), mark, name or _stem(whole), through - ) - found.extend(said) - if any(one.severity == "error" for one in found): - return Prophesied(tuple(found), None) - # After the gate rather than before it: what a flow ships is a fact about the file - # beside the source and not about the source, so saying the two have drifted apart must - # not take away the graph the source compiled to -- shipping that graph again is the - # one thing that answers the finding, and it is the one thing this would refuse. - if prophecy is not None: - found.extend(_shipped(whole, prophecy)) - return Prophesied(tuple(found), prophecy) - - -def _shipped(whole: _Whole, prophecy: Prophecy) -> list[Finding]: - """Whether the prophecy a flow ships is the one its source compiles to. - - A flowverse may ship `prophecy.pkl` beside an atlas, and that is what a run of it walks. - So a shipped prophecy which is no longer what the source says is a flow that does one - thing and reads as another -- the one thing shipping it was meant to rule out. - - Args: - whole: The parsed files. - prophecy: What the source compiles to now. - - Returns: - A `stale-prophecy` error where the two differ, and nothing where they agree, where the - flow ships none, or where what it ships is another of the atlases its file holds. - """ - from .finding import ENTRY - from .prophecy import shipped - - # Beside the entry point, which means the flow's own directory: a flow that is a single - # file has none, and what is beside such a flow is the other flows. - held = shipped(whole.entry.parent) if whole.entry.name == ENTRY else None - if held is None: - return [] - if held.prophecy is None: - return [ - _said( - "stale-prophecy", - held.at, - 0, - "the prophecy shipped here cannot be read back -- compile the atlas again, " - "or take the file away and let each run compile it", - ) - ] - if held.prophecy.name != prophecy.name: - return [] - was, now = digest(held.prophecy), digest(prophecy) - if was == now: - return [] - return [ - _said( - "stale-prophecy", - held.at, - 0, - f"the prophecy shipped here is {was} and this source compiles to {now} -- a " - "run walks the shipped one, so the flow does one thing and reads as another", - ) - ] - - -def _dynamic(read: _Read) -> list[Finding]: - """Every place one file reaches for a flow that is not an atlas. - - An atlas calls an atlas. `load` answers with a flow that may be anything -- a loop, a - branch, a week of turns -- and a prophecy with one of those in it would be a graph with - a hole where a node should be, which is the one thing a prophecy is for not having. - - Args: - read: The file. - - Returns: - A `dynamic-call` error per import of `load`, said where it is imported rather than - where it is called: a name a file has is a name a body may reach for. - """ - if not read.load_alias: - return [] - return [ - _said( - "dynamic-call", - read.where, - node.lineno, - "an atlas calls an atlas: load() answers with a flow that may be anything, " - "and what an atlas reaches another by is sub(), which is compiled into the " - "prophecy reaching for it", - ) - for node in ast.walk(read.tree) - if isinstance(node, ast.ImportFrom) - and node.module == "hmz._legacy_flows" - and any(one.name == "load" for one in node.names) - ] - - -def is_atlas(flow: str | os.PathLike[str]) -> bool: - """Whether one flow is an atlas, which is what says which reading it gets. - - Read off the entry point alone rather than off everything the flow holds: the mark that - says so is on a function in that file, and whoever is asking has a choice to make before - paying for the whole reading. - - Args: - flow: The flow: its directory, or the Python file a single-file one is. - - Returns: - Whether anything in its entry point is marked `@atlas`. False for a flow that is not - there, or will not parse -- which is a flow the other reading has plenty to say about. - """ - from .finding import ENTRY - - at = Path(flow) - entry = at / ENTRY if at.is_dir() else at - if not entry.is_file(): - return False - read = _parsed(entry) - return not isinstance(read, Finding) and any(one.atlas for one in read.marks) - - -def named_as(under: Path, inside_: str = "") -> str: - """What one atlas is called, given where its flow is and which of them was asked for. - - Args: - under: The flow's own directory, or the file a single-file flow is. - inside_: Which of the atlases the file holds was asked for, and "" for the one it - holds under its own name. - - Returns: - The name that prophecy carries, which is what a shipped one is matched against. - """ - return inside_ or (under.stem if under.is_file() else under.name) - - -def _stem(whole: _Whole) -> str: - """What the atlas a file holds under its own name is called, which is the file's.""" - from .finding import ENTRY - - at = whole.entry - return named_as(at.parent if at.name == ENTRY else at) - - -# --------------------------------------------------------------------------------------- -# What the flow's files declare, gathered across them. -# --------------------------------------------------------------------------------------- - - -class _Held(NamedTuple): - """Everything one atlas's files declare, gathered across them. - - A flow is a directory, and what it declares is spread over the files in it: the models - in one, the nodes in another, the atlas itself in the entry point. What resolves a name - in a body is therefore the whole directory rather than the file the body is in. - - Attributes: - models: The pydantic models, by name -- what a node may take and answer with. - crews: The NamedTuples, by name -- what a flow declares its agents as. - nodes: The functions marked `@mind` or `@logic`, by name. - atlases: The functions marked `@atlas`, by name, each beside the file it is in. - subs: The atlases of other files this one named, `: ` apiece. - protos: The local name of each flow-facing interface, which is how an agent reads. - fields: The local names of pydantic's `Field`, for reading whether one is required. - """ - - models: dict[str, ast.ClassDef] - crews: dict[str, ast.ClassDef] - nodes: dict[str, _Node] - atlases: dict[str, tuple[_Read, _Mark]] - subs: dict[str, str] - protos: dict[str, str] - fields: set[str] - - -def _gathered(whole: _Whole) -> _Held: - """What every file of one atlas declares, in one place. - - Args: - whole: The parsed files. - - Returns: - The declarations, by name. - """ - held = _Held({}, {}, {}, {}, {}, {}, set()) - for read in whole.read: - held.models.update(read.models) - held.crews.update(read.crews) - held.nodes.update(read.nodes) - held.subs.update(read.subs) - held.protos.update(read.proto) - held.fields.update(read.field_alias) - for mark in read.marks: - if mark.atlas: - held.atlases[mark.node.name] = (read, mark) - return held - - -# --------------------------------------------------------------------------------------- -# The entry point read: what it drives, what it is called with, what it answers with. -# --------------------------------------------------------------------------------------- - - -def _compiled( - whole: _Whole, - held: _Held, - mark: _Mark, - named: str, - through: tuple[tuple[str, str], ...], -) -> tuple[Prophecy | None, list[Finding]]: - """One atlas's entry point read, and its body walked into a prophecy. - - Args: - whole: The parsed files. - held: What those files declare, gathered across them. - mark: The atlas being compiled. - named: What to call the prophecy, which is the name the flow is asked for by. - through: The atlases this one is inside, for the supernode that reaches back. - - Returns: - The prophecy, or None where the reading refused it, and everything the reading found. - """ - where = next((one.where for one in whole.read if mark in one.marks), whole.entry) - found: list[Finding] = [] - node = mark.node - params = [*node.args.posonlyargs, *node.args.args] - if isinstance(node, ast.AsyncFunctionDef): - found.append( - _said( - "unstatic-body", - where, - node.lineno, - "an atlas is compiled rather than awaited: the body is read and the " - "graph is what runs, so there is nothing here to wait for", - ) - ) - if not _TAKES <= len(params) <= _AND_A_CONFIG: - found.append( - _said( - "unshaped-node", - where, - node.lineno, - "an atlas takes the agents and what it is called with, and after them a " - f"config for one that says it can be set up -- this takes {len(params)}", - ) - ) - return None, found - agents = _agents(params[0], held, where, found) - takes = _kind( - params[1].annotation, held, "what the atlas is called with", where, found - ) - gives = _kind(node.returns, held, "what the atlas answers with", where, found) - config = ( - _kind(params[2].annotation, held, "what an atlas is set up with", where, found) - if len(params) == _AND_A_CONFIG - else "" - ) - if config: - found.extend(_settled(config, held, where, params[2].lineno)) - answers = "" if gives == NOTHING else gives - wiring = _Wiring( - whole=whole, - held=held, - where=where, - agents=agents, - takes=takes, - gives=answers, - config=config, - names={params[0].arg: AGENTS, params[1].arg: INPUT} - | ({params[2].arg: CONFIG} if len(params) == _AND_A_CONFIG else {}), - through=(*through, (_who(where, mark.name), named)), - ) - wiring.walk(node.body) - found.extend(wiring.found) - if any(one.severity == "error" for one in found): - return None, found - return ( - Prophecy( - name=named, - takes=takes, - gives=answers, - config=config, - agents=agents, - nodes=tuple(wiring.nodes), - edges=tuple(wiring.edges), - shapes=tuple(_shapes(wiring.carried, held)), - prophecies=tuple(wiring.prophecies), - ), - found, - ) - - -def _kind( - annotation: ast.expr | None, - held: _Held, - called: str, - where: Path, - found: list[Finding], -) -> str: - """One shape an atlas's entry point declares, having said so where it declares none. - - Args: - annotation: The annotation. - held: What the flow's files declare. - called: What this place is, for a finding. - where: The file. - found: What to add a finding to. - - Returns: - The shape, and "" for an annotation that names none. - """ - shape = _shape(annotation, held) - if shape: - return shape - found.append( - _said( - "unshaped-node", - where, - annotation.lineno if annotation is not None else 0, - f"{called} is annotated {_wrote(annotation)}, which is no shape -- a model it " - f"declares, one of {', '.join(PLAIN)}, or None for nothing at all", - ) - ) - return "" - - -def _settled(config: str, held: _Held, where: Path, line: int) -> list[Finding]: - """Whether an atlas's config can be built by a run that was not set up. - - A run may be started with nothing, and the body of an atlas has no way to say what to do - about that: `config or Config()` is work, and work is what a node is for. So a run - nobody set up is handed the model's own defaults -- and a model that cannot be built out - of its defaults is one such a run has no config for at all. - - Args: - config: The model, by name. - held: What the flow's files declare. - where: The file, for a finding. - line: The line the config is declared on. - - Returns: - An `unset-config` error per field the model refuses to be built without. - """ - model = held.models.get(config) - if model is None: - return [] - short = [one.name for one in _fields(model, held) if one.required] - if not short: - return [] - return [ - _said( - "unset-config", - where, - line, - f"{config} requires {', '.join(short)}, and a run that was not set up has " - "nothing to give -- an atlas is handed its config's own defaults, so every " - "field of one has to have a default", - ) - ] - - -def _agents( - param: ast.arg, held: _Held, where: Path, found: list[Finding] -) -> tuple[str, ...]: - """What one atlas calls each of the agents it drives, read off its first parameter. - - An atlas declares its agents as a NamedTuple and not as a plain tuple of them, which is - the one thing an ordinary flow may leave unsaid: every turn in a prophecy is a node that - names the agent it drives, and a place with no name is a turn nothing can be pointed at. - - Args: - param: The entry point's first parameter. - held: What the flow's files declare. - where: The file, for a finding. - found: What to add a finding to. - - Returns: - One name per agent, in the order the flow takes them. - """ - # Read through its quoting, as every other annotation here is: a crew declared - # below the atlas that drives it is named in a string, and a name in a string is - # still the name it is. - written = _unquoted(param.annotation) - crew = held.crews.get(_root(written) if written is not None else "") - if crew is None: - found.append( - _said( - "unnamed-agents", - where, - param.lineno, - "an atlas declares its agents as a NamedTuple of them, so that every turn " - f"names the agent it drives -- {_wrote(param.annotation)} says only how " - "many there are", - ) - ) - return () - named: list[str] = [] - for one in crew.body: - if not isinstance(one, ast.AnnAssign) or not isinstance(one.target, ast.Name): - continue - if not _annotated(one.annotation, held.protos): - found.append( - _said( - "unknown-agent", - where, - one.lineno, - f"{one.target.id} is annotated {_wrote(one.annotation)}, which is not " - "an agent -- what an atlas takes first is the agents it drives and " - "nothing else", - ) - ) - continue - named.append(one.target.id) - return tuple(named) - - -# --------------------------------------------------------------------------------------- -# The body walked: one node per call, one edge per way from one to the next. -# --------------------------------------------------------------------------------------- - -#: One loose end of a body being walked: the node it leaves and what has to hold to be -#: leaving by it, where "" is the way into the prophecy and None is a node's only way out. -type _Loose = list[tuple[str, When | None]] - -#: What one node takes, as the reading of its declaration left it: the parameter's name and -#: the shape it holds, which may be one of the two agent kinds instead. -type _Takes = list[tuple[str, str]] - - -class _Declared(NamedTuple): - """What one thing a body calls is, read off wherever it is declared. - - Attributes: - kind: Which of the three kinds of node it is. - takes: Its parameters, as `(name, shape)` pairs, where the shape may be one of the two - agent kinds instead. - gives: The shape it answers with, and "" for one that answers with nothing. - rerun: Whether a run picked up inside it runs it again, or steps past it. - under: For a supernode, the prophecy it is, by name. "" for every other node. - """ - - kind: Kind - takes: _Takes - gives: str - rerun: bool - under: str = "" - - -class _Wiring: - """One atlas's body being walked into nodes and edges. - - A body is a chain of statements, and the loose ends between them are what the next - statement is wired to. A branch splits the loose ends and guards each half; a loop wires - them back to the node whose answer the loop reads; a return wires them to the end. - """ - - def __init__( - self, - *, - whole: _Whole, - held: _Held, - where: Path, - agents: tuple[str, ...], - takes: str, - gives: str, - config: str, - names: dict[str, str], - through: tuple[tuple[str, str], ...], - ) -> None: - """Holds what one body is being walked into. - - Args: - whole: The parsed files, for a supernode beside this one. - held: What the flow's files declare. - where: The file the body is in. - agents: What the atlas calls each of the agents it drives. - takes: The shape the atlas is called with. - gives: The shape it answers with, and "" for one that answers with nothing. - config: The shape it can be set up with, and "" for one that takes no setting up. - names: The entry point's own parameters, by what the body calls each. - through: The atlases this body is inside, for the supernode that reaches back. - """ - self.whole = whole - self.held = held - self.where = where - self.agents = agents - self.takes = takes - self.gives = gives - self.config = config - self.names = names - self.through = through - self.nodes: list[Node] = [] - self.edges: list[Edge] = [] - self.prophecies: list[Prophecy] = [] - self.found: list[Finding] = [] - #: What each name the body has bound holds, by shape. A name keeps the shape it was - #: first bound with: a loop binds the same name every round, and one whose shape - #: moved would be an edge that fits on the first round and not on the second. - self.bound: dict[str, str] = {} - #: How many nodes each callee has been so far, for the id the next one gets. - self.seen: dict[str, int] = {} - #: The names a refused statement would have bound. Nothing is known about what they - #: hold, and reading one is not a second mistake: it is the first one, further down. - self.spoilt: set[str] = set() - #: Every shape anything in this prophecy carries, for the shapes it is written with. - self.carried: set[str] = {one for one in (takes, gives, config) if one} - - def walk(self, body: list[ast.stmt]) -> None: - """Walks the whole of one atlas's body, and ends whatever it leaves open. - - Args: - body: The entry point's statements. - """ - for out_of, when in self._block(body, [("", None)]): - # A path that runs off the bottom of the body ends the run, and answers with - # nothing: an atlas that says it answers with something says so on every way - # out of it, which is what a `return` there is. - if self.gives: - self.found.append( - _said( - "shape-mismatch", - self.where, - body[-1].lineno if body else 0, - f"the atlas answers with {self.gives}, and this way out of it ends " - "without returning anything", - ) - ) - break - self.edges.append(Edge(out_of, "", when)) - # Only where nothing else was wrong: a body whose one statement was refused has no - # nodes *because* of that, and saying both is saying one thing twice. - if not self.nodes and not self.found: - self.found.append( - _said( - "unstatic-body", - self.where, - body[0].lineno if body else 0, - "an atlas with no nodes in it is not a graph -- a body is a call " - "apiece to the minds and logics that do the work", - ) - ) - - # -- the statements ------------------------------------------------------------- - - def _block(self, body: list[ast.stmt], loose: _Loose) -> _Loose: - """One run of statements, each wired to whatever the last one left open. - - Args: - body: The statements. - loose: The ends coming in. - - Returns: - The ends left open at the bottom, which is nothing at all after a `return`. - """ - for at, one in enumerate(body): - if _is_docstring(one, at) or isinstance(one, ast.Pass): - continue - if not loose: - self._refuse(one, "nothing here can run: the atlas ends above it") - return [] - if isinstance(one, ast.Assign) and isinstance(one.value, ast.Call): - binds, held = self._target(one), len(self.nodes) - loose = self._call(one.value, binds, loose) - if binds and len(self.nodes) == held: - # The call was refused, so the name it would have bound holds nothing: - # reading it below is this same mistake again rather than another. - self.spoilt.add(binds) - elif isinstance(one, ast.Expr) and isinstance(one.value, ast.Call): - loose = self._call(one.value, "", loose) - elif isinstance(one, ast.If): - loose = self._branch(one, loose) - elif isinstance(one, ast.While): - loose = self._loop(one, loose) - elif isinstance(one, ast.Return): - self._return(one, loose) - loose = [] - else: - self._refuse(one, f"`{_wrote(one).splitlines()[0]}` is not one of them") - return loose - - def _return(self, node: ast.Return, loose: _Loose) -> None: - """A `return`, which is where the run ends. - - Args: - node: The statement. - loose: The ends arriving at it. - """ - given = NOTHING - answers = "" - if node.value is not None: - if not isinstance(node.value, ast.Name): - self._refuse( - node, - "an atlas returns a name a node bound, whole -- a field of one is a " - "thing a logic node reads", - ) - return - # Through the same table a read goes through: what the entry point calls its - # own arguments is not what the run holds them under, so an atlas that answers - # with what it was called with names `@input` here as every other read does. - answers = self.names.get(node.value.id, node.value.id) - given = self._shape_of(Reads(answers)) or NOTHING - if given != (self.gives or NOTHING): - self.found.append( - _said( - "shape-mismatch", - self.where, - node.lineno, - f"the atlas answers with {self.gives or NOTHING} and this returns " - f"{given}", - ) - ) - for out_of, when in loose: - self.edges.append(Edge(out_of, "", when, answers)) - - def _branch(self, node: ast.If, loose: _Loose) -> _Loose: - """An `if`, which is the several ways out of the node above it. - - Args: - node: The statement. - loose: The ends arriving at it, each of which the branch guards. - - Returns: - The ends both arms left open. - """ - read = self._branched(node.test, loose) - if read is None: - return loose - said, truth = read - taken: _Loose = [(out_of, When(*said, truth)) for out_of, _ in loose] - otherwise: _Loose = [(out_of, When(*said, not truth)) for out_of, _ in loose] - # One arm at a time, each starting from what was bound above it: a name one arm - # binds is a name the other path arrives without, and a node below the branch that - # read it would be handed nothing on that path. - was = dict(self.bound) - held = self._block(node.body, taken) - then, self.bound = self.bound, dict(was) - other = self._block(node.orelse, otherwise) - # And below the branch, only what both arms bound and bound the same: everything - # else is a name that holds something on one way here and nothing on the other. - self.bound = { - name: shape for name, shape in then.items() if self.bound.get(name) == shape - } - return [*held, *other] - - def _loop(self, node: ast.While, loose: _Loose) -> _Loose: - """A `while`, which is a branch with an edge back to the node it reads. - - The node above the loop is its head: it answers the name the test reads, the body - runs while that holds, and the body's last node wires back to the head -- so the - head answers again with whatever the round changed, which is what ends the loop. - - Args: - node: The statement. - loose: The ends arriving at it, which is one and no more. - - Returns: - The one end the loop leaves open, guarded by the test not holding. - """ - if node.orelse: - self._refuse( - node, "a `while` in an atlas has no `else`: the loop is its edges" - ) - return loose - if len(loose) != 1: - self._refuse( - node, - "a loop leaves the one node it reads again each round -- put a logic node " - "between the branch above and this, so the loop has a head", - ) - return loose - read = self._branched(node.test, loose) - if read is None: - return loose - head = loose[0][0] - said, truth = read - # And the node above it is the one it reads again: the loop's edge goes back there, - # so a `while` whose guard that node does not answer is a guard no round can change - # -- a loop with no way out, compiled from a body that reads as though it had one. - above = self._above(head) - if above is None or above.binds != said.reads: - self._refuse( - node, - f"a loop reads again the node above it, and this reads {_names(said)}, " - f"which {head or 'nothing yet'} does not answer -- put the node that " - "answers it directly above the loop, so each round asks it again", - ) - return loose - opens: _Loose = [(head, When(*said, truth))] - inside = self._block(node.body, opens) - self._twice(node, head, inside) - for out_of, when in inside: - self.edges.append(Edge(out_of, head, when)) - self._endless(node, head) - ends: _Loose = [(head, When(*said, not truth))] - return ends - - def _twice(self, node: ast.While, head: str, inside: _Loose) -> None: - """Whether a loop's body ends with the very node the loop reads again. - - Writing the head again at the bottom of the body is the natural Python and the wrong - graph: the edge back goes to the head, so the head answers again anyway. The body's - copy would run first, its answer would be thrown away, and a node with an effect -- - a line written, a message sent -- would have it twice a round with nothing said. - - Args: - node: The loop. - head: The node it reads again, by node id. - inside: The ends the body left open. - """ - above = self._above(head) - if above is None: - return - for out_of, _ in inside: - last = self._above(out_of) - if last is not None and last.calls == above.calls: - self.found.append( - _said( - "twice-round", - self.where, - node.lineno, - f"the body of this loop ends with {above.calls}, which is what the " - "loop reads again each round -- so it would run twice a round and " - "the body's answer be thrown away; take it out of the body", - ) - ) - return - - def _endless(self, node: ast.While, head: str) -> None: - """Whether one loop's head can ever answer differently, which is whether it ends. - - Args: - node: The loop. - head: The node it reads again each round, by node id. - """ - wrote = { - one.id - for said in ast.walk(node) - if isinstance(said, ast.Assign) - for one in said.targets - if isinstance(one, ast.Name) - } - reading = self._above(head) - if reading is not None and not wrote & {one.reads for one in reading.takes}: - self.found.append( - _said( - "dead-loop", - self.where, - node.lineno, - f"nothing in this loop changes what {head} reads, so it answers the " - "same thing every round and the loop never ends", - ) - ) - - def _branched(self, test: ast.expr, loose: _Loose) -> tuple[Reads, bool] | None: - """What one branch reads, and whether the nodes above it may be branched on. - - Args: - test: The `if` or `while` test. - loose: The ends arriving at the branch. - - Returns: - What the branch reads, and whether the first way out is the one taken when that - reads as true. None where the branch is refused. - """ - # `not` and nothing else: every other unary operator is work -- `~x` is falsy where - # `x` is truthy -- and one read as though it were the name under it would be a graph - # that branches the other way from the body it was compiled from. - said: ast.expr = ( - test.operand if isinstance(test, ast.UnaryOp) and _is_not(test) else test - ) - truth = not _is_not(test) - read = self._reads(said) - if read is None: - self._refuse( - test, - f"a branch reads a name a node bound, or one field of it -- `{_wrote(test)}`" - " is work, and work is what a logic node is for", - ) - return None - shape = self._shape_of(read) - if shape is None: - if read.reads not in self.spoilt: - self.found.append( - _said( - "unbound-read", - self.where, - test.lineno, - f"nothing here has bound {_names(read)}", - ) - ) - return None - if not shape: - self.found.append( - _said( - "shape-mismatch", - self.where, - test.lineno, - f"{_names(read)} holds nothing -- the node that bound it answers with " - "nothing at all, so a branch reading it is one way out taken every " - "round and one taken never", - ) - ) - return None - for out_of, when in loose: - if when is not None: - self._refuse( - test, - "a branch follows a node and not another branch -- an `elif`, or an " - "arm with nothing in it, is two decisions carried on one edge; put a " - "logic node between them so each way out belongs to what decided it", - ) - return None - above = self._above(out_of) - if above is None: - self._refuse( - test, "a branch follows a node, and nothing has run here yet" - ) - return None - if above.kind == "mind": - self.found.append( - _said( - "branching-mind", - self.where, - test.lineno, - f"{out_of} is a turn, and a turn has one way out -- read what it " - "answered with a logic node, and branch on that", - ) - ) - return None - return read, truth - - def _above(self, at: str) -> Node | None: - """The node of that id, or None for the way into the prophecy.""" - return next((one for one in self.nodes if one.at == at), None) - - def _target(self, node: ast.Assign) -> str: - """The one name an assignment binds, having said so where it binds anything else.""" - if len(node.targets) == 1 and isinstance(node.targets[0], ast.Name): - return node.targets[0].id - self._refuse( - node, - "a node binds one name -- nothing here is unpacked, and nothing bound twice", - ) - return "" - - # -- one node ------------------------------------------------------------------- - - def _call(self, call: ast.Call, binds: str, loose: _Loose) -> _Loose: - """One call, which is one node of the prophecy. - - Args: - call: The call. - binds: The name its answer is bound to, and "" for one nothing takes. - loose: The ends arriving at it. - - Returns: - The one end it leaves open. - """ - if not isinstance(call.func, ast.Name): - self._refuse(call, "a node is one call to a name this flow declares") - return loose - called = call.func.id - if call.keywords or any(isinstance(one, ast.Starred) for one in call.args): - self._refuse( - call, - "a node is handed its arguments in order, by name -- no keywords and " - "nothing unpacked, so that what flows along each edge is one thing", - ) - return loose - if binds in self.names: - # What the entry point calls its own arguments is what the run holds the task, - # the agents and the config under, and a body binding one of those names would - # be a node whose answer nothing below it could read: every read of that name - # goes on answering with what the atlas was called with. - self._refuse( - call, - f"{binds} is what this atlas was called with -- bind the answer to a name " - "of its own, so that what reads it reads the node and not the run", - ) - return loose - declared = self._declared(call, called) - if declared is None: - return loose - kind, takes, gives, rerun, under = declared - reads = self._arguments(call, called, takes) - if reads is None: - return loose - if not rerun and gives: - self.found.append( - _said( - "skipped-answer", - self.where, - call.lineno, - f"{called} is stepped past when a run is picked up inside it, and " - f"answers with {gives} -- a node a run steps past has no answer for " - "what comes next, so such a node answers with nothing", - ) - ) - at = self._id(called) - self.nodes.append( - Node( - at=at, - kind=kind, - calls=under or called, - takes=tuple(reads), - binds=binds, - gives=gives, - rerun=rerun, - under=under, - ) - ) - self.carried.update( - shape for _, shape in takes if shape not in (ONE_AGENT, THE_AGENTS) - ) - if gives: - self.carried.add(gives) - if binds: - self._binds(call, binds, gives) - for out_of, when in loose: - self.edges.append(Edge(out_of, at, when)) - return [(at, None)] - - def _declared(self, call: ast.Call, called: str) -> _Declared | None: - """What the thing one call names is, and what it takes and answers with. - - Args: - call: The call, for a finding. - called: The name it calls. - - Returns: - Its declaration, or None where the name is not something an atlas may call. - """ - held = self.held.nodes.get(called) - if held is not None: - return self._noded(call, called, held) - if called in self.held.atlases or called in self.held.subs: - return self._supernode(call, called) - self._refuse( - call, - f"{called} is not a node: an atlas calls what it marked @mind or @logic, an " - "atlas beside it, or one it named with sub()", - ) - return None - - def _noded(self, call: ast.Call, called: str, held: _Node) -> _Declared | None: - """One `@mind` or `@logic` node, read off the function that declares it. - - Args: - call: The call, for a finding. - called: The name it calls. - held: What the mark said. - - Returns: - Its declaration, or None where the function's shapes cannot be read. - """ - if isinstance(held.node, ast.AsyncFunctionDef): - self.found.append( - _said( - "unstatic-body", - self.where, - call.lineno, - f"{called} is a coroutine, and the walk over a prophecy does not " - "await -- a node answers with a shape, and one answering with a " - "coroutine would hand the next node something no model is built from", - ) - ) - return None - params = [*held.node.args.posonlyargs, *held.node.args.args] - takes: _Takes = [] - for at, one in enumerate(params): - agent = bool(_annotated(one.annotation, self.held.protos)) - if agent and at == 0 and held.kind == "mind": - takes.append((one.arg, ONE_AGENT)) - continue - if agent: - self.found.append( - _said( - "unagented-node", - self.where, - call.lineno, - f"{called} takes an agent as {one.arg} -- a mind takes one and it " - "is the first thing it takes, and a logic takes none at all", - ) - ) - return None - shape = _shape(one.annotation, self.held) - if not shape or shape == NOTHING: - self.found.append( - _said( - "unshaped-node", - self.where, - call.lineno, - f"{called} takes {one.arg} annotated {_wrote(one.annotation)}, " - "which is no shape -- a node takes a model this flow declares, or " - f"one of {', '.join(PLAIN)}", - ) - ) - return None - takes.append((one.arg, shape)) - if held.kind == "mind" and not (takes and takes[0][1] == ONE_AGENT): - self.found.append( - _said( - "unagented-node", - self.where, - call.lineno, - f"{called} is a turn and takes no agent -- what a mind takes first is " - "the agent it drives", - ) - ) - return None - gives = _shape(held.node.returns, self.held) - if not gives: - self.found.append( - _said( - "unshaped-node", - self.where, - call.lineno, - f"{called} answers with {_wrote(held.node.returns)}, which is no " - f"shape -- a model this flow declares, one of {', '.join(PLAIN)}, or " - "None for nothing at all", - ) - ) - return None - kind: Kind = "mind" if held.kind == "mind" else "logic" - return _Declared( - kind, takes, "" if gives == NOTHING else gives, rerun=held.rerun - ) - - def _supernode(self, call: ast.Call, called: str) -> _Declared | None: - """One supernode: a whole atlas, compiled into the prophecy reaching for it. - - Args: - call: The call, for a finding. - called: The name it calls. - - Returns: - Its declaration, or None where the atlas under it did not compile. - """ - named = self.held.subs.get(called, called) - under = next((one for one in self.prophecies if one.name == named), None) - if under is None: - under = self._under(call, called, named) - if under is None: - return None - self.prophecies.append(under) - if under.config: - self.found.append( - _said( - "unstatic-body", - self.where, - call.lineno, - f"{named} says it can be set up, and a supernode is a node: what is " - "set up is the run, so an atlas that takes a config is one to start " - "rather than one to reach for", - ) - ) - return None - short = set(under.agents) - set(self.agents) - if short: - self.found.append( - _said( - "unknown-agent", - self.where, - call.lineno, - f"{named} drives {', '.join(sorted(short))}, which this atlas does " - "not -- a supernode is handed the agents of the run around it, by the " - "names it calls them", - ) - ) - return None - return _Declared( - "atlas", - [("agents", THE_AGENTS), ("said", under.takes)], - under.gives, - rerun=True, - under=named, - ) - - def _under(self, call: ast.Call, called: str, named: str) -> Prophecy | None: - """The prophecy one supernode is, compiled where it stands. - - Args: - call: The call, for a finding. - called: The name it calls. - named: The atlas it is, as this flow named it. - - Returns: - The prophecy, or None where compiling it found an error. - """ - beside = self.held.atlases.get(called) - if beside is not None: - read, mark = beside - if self._circular(call, named, _who(read.where, mark.name)): - return None - made, found = _compiled(self.whole, self.held, mark, named, self.through) - self.found.extend(found) - return made - from .finding import ENTRY, find, inside - - at = Path(find(named)) - if self._circular(call, named, _who(at, inside(named))): - return None - held = prophesied( - at.parent if at.name == ENTRY else at, - name=inside(named), - through=self.through, - ) - self.found.extend( - one - for one in held.findings - # A supernode is compiled where it is reached for, so what its own reading - # found is said once. The name it was reached by is what places it. - if one.severity == "error" or one.code != "unsaid-flow" - ) - if held.prophecy is None and not any( - one.severity == "error" for one in held.findings - ): - self.found.append( - _said( - "not-an-atlas", - self.where, - call.lineno, - f"{named} is not an atlas -- an atlas calls an atlas, and reaches an " - "ordinary flow through nothing at all", - ) - ) - return None - return None if held.prophecy is None else held.prophecy._replace(name=named) - - def _circular(self, call: ast.Call, named: str, who: str) -> bool: - """Whether one supernode reaches back into an atlas already being compiled. - - Asked of where the atlas is and what it is called there rather than of the name the - body wrote: one atlas is reached as `deeper` beside it and as `epic:deeper` from - anywhere else, and a check that compared spellings would follow that forever. - - Args: - call: The call, for a finding. - named: The atlas, as this body named it. - who: Which atlas it is: where it is declared, and what it is called there. - - Returns: - Whether it does, having said so where it does. - """ - if who not in {one for one, _ in self.through}: - return False - self.found.append( - _said( - "circular-atlas", - self.where, - call.lineno, - f"{named} is being compiled already -- a supernode is one graph inside " - f"another, and {' inside '.join((*(said for _, said in self.through), named))}" - " has no bottom", - ) - ) - return True - - def _arguments( - self, call: ast.Call, called: str, takes: _Takes - ) -> list[Reads] | None: - """Where each of one node's arguments comes from, held to what it takes. - - Args: - call: The call. - called: What it calls, for a finding. - takes: The node's parameters as `(name, shape)` pairs. - - Returns: - One per argument, or None where the call does not fit what it calls. - """ - if len(call.args) != len(takes): - self.found.append( - _said( - "shape-mismatch", - self.where, - call.lineno, - f"{called} takes {len(takes)} and is handed {len(call.args)}", - ) - ) - return None - reads: list[Reads] = [] - for one, (param, shape) in zip(call.args, takes, strict=True): - read = self._reads(one) - if read is None: - self._refuse( - one, - "an argument is a name a node bound or a field of one -- " - f"`{_wrote(one)}` is work, and work is what a node is for", - ) - return None - if not self._fits(call, called, param, shape, read): - return None - reads.append(read) - return reads - - def _fits( - self, call: ast.Call, called: str, param: str, shape: str, read: Reads - ) -> bool: - """Whether what flows into one parameter is what that parameter takes. - - Args: - call: The call, for a finding. - called: What it calls. - param: The parameter's name. - shape: What it takes, or one of the two agent kinds. - read: The name and field being handed to it. - - Returns: - Whether it fits, having said why where it does not. - """ - if shape in (ONE_AGENT, THE_AGENTS): - return self._agented(call, called, param, shape, read) - given = self._shape_of(read) - if given is None: - if read.reads not in self.spoilt: - self.found.append( - _said( - "unbound-read", - self.where, - call.lineno, - f"nothing here has bound {_names(read)}", - ) - ) - return False - if not given or not _same(given, shape, self.held): - self.found.append( - _said( - "shape-mismatch", - self.where, - call.lineno, - f"{called} takes {param}: {shape}, and {_names(read)} is " - f"{given or _wrote(None)}", - ) - ) - return False - return True - - def _agented( - self, call: ast.Call, called: str, param: str, shape: str, read: Reads - ) -> bool: - """Whether what flows into an agent's place is one of the run's own agents. - - Args: - call: The call, for a finding. - called: What it calls. - param: The parameter's name. - shape: Which of the two agent kinds it is. - read: The name and field being handed to it. - - Returns: - Whether it fits, having said why where it does not. - """ - reads, field = read - one = shape == ONE_AGENT - if reads != AGENTS or bool(field) != one: - self.found.append( - _said( - "unagented-node", - self.where, - call.lineno, - f"{called} takes {'one of the agents' if one else 'the agents'} as " - f"{param}, and is handed {_names(read)}", - ) - ) - return False - if one and field not in self.agents: - self.found.append( - _said( - "unknown-agent", - self.where, - call.lineno, - f"this atlas drives {', '.join(self.agents) or 'nothing'}, and " - f"{_names(read)} is none of them", - ) - ) - return False - return True - - # -- the names a body binds and reads -------------------------------------------- - - def _reads(self, node: ast.expr) -> Reads | None: - """One name a body reads, and the field read off it. - - Args: - node: The expression. - - Returns: - `(the name, the field)`, the field being "" for the whole of it -- or None where - this is not a name at all. - """ - if isinstance(node, ast.Name): - return Reads(self.names.get(node.id, node.id)) - if isinstance(node, ast.Attribute) and isinstance(node.value, ast.Name): - return Reads(self.names.get(node.value.id, node.value.id), node.attr) - return None - - def _shape_of(self, read: Reads) -> str | None: - """What one name, or one field of it, holds. - - Args: - read: The name and the field read off it. - - Returns: - The shape; "" for something bound whose field is no shape an edge may carry; and - None for a name nothing here has bound. - """ - reads, field = read - held = {INPUT: self.takes, CONFIG: self.config}.get(reads) or self.bound.get( - reads - ) - if held is None or reads == AGENTS: - return None - if not field: - return held - model = self.held.models.get(held) - if model is None: - return None - return next( - ( - _shape(one.annotation, self.held) or "" - for one in model.body - if isinstance(one, ast.AnnAssign) - and isinstance(one.target, ast.Name) - and one.target.id == field - ), - None, - ) - - def _binds(self, call: ast.Call, binds: str, gives: str) -> None: - """Binds one name to what the node answered with, once and for the whole body. - - Args: - call: The call, for a finding. - binds: The name. - gives: The shape it now holds. - """ - held = self.bound.get(binds) - if held is not None and held != gives: - self.found.append( - _said( - "shape-mismatch", - self.where, - call.lineno, - f"{binds} is {held} above and {gives or NOTHING} here -- a name keeps " - "the shape it was bound with, so that an edge which fits on the first " - "round fits on every round", - ) - ) - return - self.bound[binds] = gives - - def _id(self, called: str) -> str: - """The node id one more call to the same thing gets.""" - self.seen[called] = at = self.seen.get(called, 0) + 1 - return called if at == 1 else f"{called}:{at}" - - def _refuse(self, node: ast.stmt | ast.expr, said: str) -> None: - """Says one thing the body holds that the subset an atlas is written in does not. - - Whatever the refused statement would have bound is remembered as spoilt, so that - reading it further down is not reported as a mistake of its own: one thing wrong in - a body is one finding, and a reader given four for it has three to work out are - consequences. - - Args: - node: The statement or expression being refused. - said: What is wrong with it. - """ - self.found.append(_said("unstatic-body", self.where, node.lineno, said)) - self.spoilt.update( - one.id - for held in ast.walk(node) - if isinstance(held, ast.Assign) - for one in held.targets - if isinstance(one, ast.Name) - ) - - -# --------------------------------------------------------------------------------------- -# The shapes: what may flow along an edge, and whether one fits another. -# --------------------------------------------------------------------------------------- - - -def _shape(annotation: ast.expr | None, held: _Held) -> str | None: - """What one annotation says flows there, by shape name. - - Args: - annotation: The annotation, which may be missing. - held: What the flow's files declare. - - Returns: - The model's name, one of :data:`PLAIN`, `None` for a place that carries nothing, or - None where the annotation is no shape at all. `X | None` is `X`: a shape that may be - missing is the shape, and whether it is there is what a branch reads. - """ - if annotation is None: - return None - if isinstance(annotation, ast.Constant) and annotation.value is None: - return NOTHING - # A quoted annotation is the annotation: a flow written under `from __future__ import - # annotations` and one written without it declare the same node. - said = _unquoted(annotation) - if said is None: - return None - if said is not annotation: - return _shape(said, held) - annotation = said - if isinstance(annotation, ast.Constant): - return None - if isinstance(annotation, ast.BinOp) and isinstance(annotation.op, ast.BitOr): - sides = [_shape(annotation.left, held), _shape(annotation.right, held)] - said = [one for one in sides if one is not None and one != NOTHING] - return said[0] if len(said) == 1 and None not in sides else None - if isinstance(annotation, ast.Subscript) and _tip(annotation.value) == "Annotated": - return _shape(_elements(annotation.slice)[0], held) - if not isinstance(annotation, ast.Name): - return None - if annotation.id in held.models or annotation.id in PLAIN: - return annotation.id - return NOTHING if annotation.id == NOTHING else None - - -def _same(given: str, wanted: str, held: _Held) -> bool: - """Whether what one node answers with is what the next one takes. - - The name where both are the same shape, and the fields where they are not: a model that - holds every field another requires, at the same shape apiece, is a model that model can - be built from -- which is what an edge between two of them means. - - Args: - given: The shape flowing in. - wanted: The shape the far end takes. - held: What the flow's files declare. - - Returns: - Whether it fits. - """ - if given == wanted: - return True - one, two = held.models.get(given), held.models.get(wanted) - if one is None or two is None: - return False - holds = {field.name: field.shape for field in _fields(one, held)} - return all( - holds.get(field.name) == field.shape - for field in _fields(two, held) - if field.required - ) - - -def _fields(model: ast.ClassDef, held: _Held) -> list[Field]: - """Every field one model declares, and whether it refuses to be built without it. - - Args: - model: The class. - held: What the flow's files declare, for a base declared beside it. - - Returns: - One per field, the bases' first: a model is what it inherits and what it adds. - """ - said: list[Field] = [] - for base in model.bases: - beside = held.models.get(_root(base)) - if beside is not None and beside is not model: - said.extend(_fields(beside, held)) - for one in model.body: - if not isinstance(one, ast.AnnAssign) or not isinstance(one.target, ast.Name): - continue - name = one.target.id - if name.startswith("_") or _root(one.annotation) == "ClassVar": - continue - said = [was for was in said if was.name != name] - said.append( - Field(name, _wrote(one.annotation), required=_required(one.value, held)) - ) - return said - - -def _required(value: ast.expr | None, held: _Held) -> bool: - """Whether a field with that default refuses to be built without being given one. - - Args: - value: What the field was declared with, and None where it was declared with nothing. - held: What the flow's files declare, for what each of them calls pydantic's `Field`. - - Returns: - Whether a model of it cannot be built without being handed one. - """ - if value is None: - return True - # By what it is called at the tip: `Field(...)` is what a flow writes, and - # `pydantic.Field(...)` is the same call reached the other way -- one read at the - # root would be the module's name and would read every field as one with a default. - if isinstance(value, ast.Call) and ( - _root(value.func) in held.fields or _tip(value.func) == "Field" - ): - named = {one.arg for one in value.keywords} - return not (value.args or named & {"default", "default_factory"}) - return False - - -def _shapes(carried: set[str], held: _Held) -> list[Shape]: - """Every shape one prophecy carries, written out with the fields each holds. - - Args: - carried: The shape names, as the compiling gathered them. - held: What the flow's files declare. - - Returns: - One per shape, in name order. A plain kind has no fields, having none to have -- and - nor has a model another flow declares, which a supernode's edges name and this flow - cannot read. - """ - return [ - Shape(name, tuple(_fields(model, held)) if model is not None else ()) - for name in sorted(carried) - if name and name != NOTHING - for model in (held.models.get(name),) - ] - - -# --------------------------------------------------------------------------------------- -# The small readings the rules above are written in terms of. -# --------------------------------------------------------------------------------------- - - -def _who(where: Path, name: str) -> str: - """Which atlas one name resolves to: where it is declared, and what it is called there. - - Args: - where: The file it is declared in. - name: What the mark called it inside that file. - - Returns: - The two, as one string -- what a supernode reaching back into a compiling is caught by. - """ - return f"{where.resolve()}::{name}" - - -def _said(code: str, where: Path, line: int, why: str) -> Finding: - """One finding, which for an atlas is always a reason it did not compile.""" - return Finding(code, "error", where, line, why) - - -def _is_docstring(node: ast.stmt, at: int) -> bool: - """Whether one statement is the docstring a body opens with.""" - return ( - at == 0 - and isinstance(node, ast.Expr) - and isinstance(node.value, ast.Constant) - and isinstance(node.value.value, str) - ) - - -def _is_not(node: ast.expr) -> bool: - """Whether one test is `not` something, which is the branch's other way out.""" - return isinstance(node, ast.UnaryOp) and isinstance(node.op, ast.Not) - - -def _names(read: Reads) -> str: - """How one name and the field read off it read in a finding.""" - reads, field = read - said = { - AGENTS: "the agents", - INPUT: "what the atlas was called with", - CONFIG: "the config", - }.get(reads, reads) - return f"{said}.{field}" if field else said - - -def _wrote(node: ast.expr | ast.stmt | None) -> str: - """One piece of a body as it was written, for a finding to quote back.""" - return "nothing" if node is None else ast.unparse(node) diff --git a/src/hmz/runtime/flowing/proving.py b/src/hmz/runtime/flowing/proving.py deleted file mode 100644 index 8f8cd578..00000000 --- a/src/hmz/runtime/flowing/proving.py +++ /dev/null @@ -1,755 +0,0 @@ -"""A flow driven by stubs against a clock: the reading only running the file can give. - -:mod:`hmz.runtime.flowing.checking` reads a flow without running it, and some of what a flow is only -running can show -- the annotation built at runtime, the config model declared in a helper, -the loop that looks bounded and is not. This is that second reading. The flow is loaded and -driven for real, in a subprocess of its own, by agents that are stubs: every turn lands at -once, answers deterministically, and costs what the scenario says a turn costs -- so a flow -that declared an allowance of its own walks to the end of it in milliseconds, and what is -being proved is the flow's own shape rather than any model's mood. - -The scenarios are the questions worth asking of a loop. `NEVER_DONE` is the reviewer that -never says the work is done: a flow with a bound of its own still ends, and one without is -caught by the turn cap or killed by the clock -- which is the executable proof that a run of -it can end. `ALWAYS_DONE` is the shortest road through. `SILENT` answers every turn with -nothing, which is what a turn that failed answers, so a flow that reads a field off an -unguarded answer falls over here rather than at hour three. - -A subprocess per scenario, because loading a flow means running its file: whatever it does as -it is read -- imports, prints, mistakes -- happens in a process built to be killed, the parent -holds the clock, and nothing of the flow outlives its own proof. -""" - -from __future__ import annotations - -import inspect -import json -import subprocess -import sys -import tempfile -from pathlib import Path -from typing import ( - TYPE_CHECKING, - Any, - ClassVar, - Literal, - NamedTuple, - cast, - get_args, - get_origin, -) - -from .checking import Finding - -if TYPE_CHECKING: - import os - from collections.abc import Iterator, Mapping, Sequence - - from pydantic import BaseModel - - from hmz.coganchor.agents import AgentBase, Event - from hmz.coganchor.agents.allowance import Allowance, Ledger - - from .driving import Place - -__all__ = [ - "ALWAYS_DONE", - "NEVER_DONE", - "SILENT", - "Outcome", - "Proof", - "Scenario", - "proved", -] - - -class Scenario(NamedTuple): - """One way the world answers a flow, held constant for the length of a proof. - - Attributes: - name: What the scenario is called, which is what its outcome is filed under. - verdict: What every boolean field of a shaped answer says -- False is the reviewer - that never says done -- or None for a turn that answers with nothing at all, which - is what a failed turn answers. - answer: What a plain turn answers, and what every string field of a shaped one says. - climb: What each turn adds to what the agent has spent, in output tokens, so that a - flow declaring an allowance of its own walks to the end of it in a handful of - turns rather than in the hundred thousand a real run would take. - turns: How many turns the flow may take before it is read as one that does not stop. - tick: How much wall clock each turn is worth inside the proof, in seconds. A proof - sleeps for free and a stub answers at once, so real time never moves in one -- and a - flow declaring an allowance in hours would be driven to the turn cap and reported as - one that does not stop, which is a false failure about the very shape the checker now - blesses. A minute a turn walks six declared hours in a few hundred. - seconds: How long the scenario's process may live before the clock kills it. Real - seconds, this one, being the parent's patience rather than the proof's own clock. - """ - - name: str - verdict: bool | None - answer: str - climb: float = 100_000.0 - turns: int = 200 - tick: float = 60.0 - seconds: float = 60.0 - - -#: The reviewer that never says the work is done. A flow with a bound of its own -- a cap on -#: the rounds, a clock, an allowance it declared -- still ends here, and one that waits -#: forever on a verdict is caught by the turn cap: the executable proof that a run of it can -#: end. What a real run has underneath all of them is the allowance somebody set when they -#: started it, which is not a thing a proof of the flow can stand on: the flow is on trial, -#: and a run stopped for running out of money did not end because of anything the flow did. -NEVER_DONE = Scenario("never-done", verdict=False, answer="did some of it") - -#: The shortest road through: every verdict is yes, so what is proved is that the flow can -#: end the way it means to. -ALWAYS_DONE = Scenario("always-done", verdict=True, answer="did it") - -#: Every turn answers with nothing, which is what a failed turn answers: a flow that reads -#: a field off an answer nobody guarded falls over here rather than hours into a run. -SILENT = Scenario("silent", verdict=None, answer="") - - -class Outcome(NamedTuple): - """How one scenario ended. - - Attributes: - scenario: Which scenario it was. - finished: Whether the flow ended on its own -- returned, or raised what it meant to. - turns: How many turns it took, as far as that was counted. - said: Why it did not finish, for one that did not: the clock, the turn cap, or the - tail of what it raised. "" for one that did. - """ - - scenario: str - finished: bool - turns: int - said: str - - -class Proof(NamedTuple): - """What driving one flow against the scenarios showed. - - Attributes: - findings: What loading it refused or the live reading found, in the same shape the - static reading answers with -- `refused-load` for a flow `driving.py` would not - take, and the config findings only the declared model itself can show. - outcomes: One per scenario, in the order they were asked. - """ - - findings: tuple[Finding, ...] - outcomes: tuple[Outcome, ...] - - -#: How long the load-only proof is given, there being no scenario to say. -_PATIENCE = 60.0 - -#: What the stubbed flow is driven with. Constant, so a proof is a proof of the flow: what -#: the task says cannot matter to agents that answer the same thing whatever they are told. -_TASK = "the task this proof drives the flow on" - - -def proved( - flow: str | os.PathLike[str], - *, - name: str = "", - config: Mapping[str, object] | None = None, - scenarios: tuple[Scenario, ...] = (NEVER_DONE, ALWAYS_DONE), -) -> Proof: - """Loads a flow in a subprocess and drives it with stubs, once per scenario. - - Args: - flow: The flow: its directory, its file, or the name `-f` takes. - name: Which of the flows the file holds, or "" for the one it holds under its own - name -- the half after the colon, for whoever has it separately. - config: What to set the flow up with, read back through the flow's own model exactly - as a run of it would, or None for a flow left to its defaults. - scenarios: The worlds to drive it against, each in a process of its own. Empty proves - only that it loads: the flow is declared and its config model read, and nothing - takes a turn. - - Returns: - The findings and one outcome per scenario. A finding is something to fix; an outcome - that did not finish is a flow that could not end in that world, said with why. - """ - from .finding import find, inside - - # Resolved here, where names still mean what the caller meant: the child runs in a - # scratch directory of its own, against which a relative path names nothing. - at = find(str(flow)) - wanted = name or inside(str(flow)) - where = Path(at) - findings: list[Finding] = [] - outcomes: list[Outcome] = [] - seen: set[tuple[str, str]] = set() - asked: tuple[Scenario | None, ...] = scenarios or (None,) - for scenario in asked: - told = _asked(at, wanted, config, scenario) - if isinstance(told, Outcome): - outcomes.append(told) - continue - for one in told.get("findings", ()): - key = (str(one["code"]), str(one["said"])) - if key not in seen: - seen.add(key) - severity: Literal["error", "warning"] = ( - "error" if one["severity"] == "error" else "warning" - ) - findings.append(Finding(key[0], severity, where, 0, key[1])) - refused = told.get("refused") - if refused is not None: - key = ("refused-load", str(refused)) - if key not in seen: - seen.add(key) - findings.append(Finding(key[0], "error", where, 0, key[1])) - if scenario is not None: - outcomes.append( - Outcome( - scenario.name, - finished=False, - turns=0, - said="nothing ran: the flow could not be loaded", - ) - ) - continue - if scenario is not None: - outcomes.append( - Outcome( - scenario.name, - finished=bool(told.get("finished")), - turns=int(told.get("turns", 0)), - said=str(told.get("said", "")), - ) - ) - return Proof(tuple(findings), tuple(outcomes)) - - -def _asked( - flow: str, - name: str, - config: Mapping[str, object] | None, - scenario: Scenario | None, -) -> dict[str, Any] | Outcome: - """One scenario, asked of a child process holding the clock over it. - - Args: - flow: The flow, as :func:`proved` was given it. - name: Which of the file's flows. - config: What to set it up with, or None. - scenario: The world to drive it in, or None to only load it. - - Returns: - What the child answered, or the outcome of a child that could not answer: one the - clock killed, or one that died without saying why in the one line this reads. - """ - called = scenario.name if scenario is not None else "" - spec = json.dumps( - { - "name": name, - "config": dict(config) if config is not None else None, - "scenario": scenario._asdict() if scenario is not None else None, - } - ) - patience = scenario.seconds if scenario is not None else _PATIENCE - # A scratch directory to work in, taken away with the process: what a flow writes while - # it is being proved is part of the proof, not part of anybody's repository. - with tempfile.TemporaryDirectory(prefix="hmz-proving-") as scratch: - try: - done = subprocess.run( - [sys.executable, "-m", "hmz.runtime.flowing.proving", flow, spec], - capture_output=True, - text=True, - check=False, - timeout=patience, - cwd=scratch, - ) - except subprocess.TimeoutExpired: - return Outcome( - called, - finished=False, - turns=0, - said=f"still running after {patience:g}s -- nothing inside the flow " - "ended it, so the clock did", - ) - # The last line that is the child's: a flow prints whatever it prints, so the answer is - # found from the end rather than trusted to be alone. - for line in reversed(done.stdout.splitlines()): - try: - held = json.loads(line) - except ValueError: - continue - if isinstance(held, dict) and "proving" in held: - return cast("dict[str, Any]", held["proving"]) - tail = "\n".join(done.stderr.strip().splitlines()[-3:]) - return Outcome( - called, - finished=False, - turns=0, - said=f"the flow's process ended without answering -- {tail or 'and said nothing'}", - ) - - -# --------------------------------------------------------------------------------------- -# The child: loads the flow, builds the stubs, and drives it. Run as -# `-m hmz.runtime.flowing.proving` with the flow and the scenario as its two arguments, and -# answers with one JSON line. -# --------------------------------------------------------------------------------------- - - -def _rested(seconds: float) -> None: - """A sleep that has already happened, which is what the stubs' world does with rests. - - Args: - seconds: How long the flow meant to wait, which the proof does not. - """ - del seconds - - -async def _rested_for(seconds: float, result: Any = None) -> Any: - """The same for a flow that rests the async way, which `asyncio.sleep` is. - - Args: - seconds: How long the flow meant to wait, which the proof does not. - result: What `asyncio.sleep` answers with, which it hands back untouched. - - Returns: - That same thing, at once. - """ - del seconds - return result - - -def _allowed( - driven: list[AgentBase], - declared: Allowance | None, - scenario: Scenario, - steps: _Steps, -) -> Ledger | None: - """Holds the stubs to the allowance the flow itself declared, and to no other. - - The flow's own or none at all. A real run is held to whatever the person starting it - set, and a proof that stood on one of those would be proving something about a number - somebody typed rather than about the flow -- a loop that never ends would pass, having - been stopped by money. What the flow declared is the flow's, so it is on trial with the - rest of it: a flow saying `@flow(budget=Allowance(tokens=10))` is a flow claiming that - ten million output tokens is where a run of it ends, and this is that claim being tried. - - Two of the three dimensions need the proof's world to move for them. Tokens move already: - a stub turn spends `climb` of them. The clock does not -- a proof sleeps for free and a - stub answers at once -- so it is driven off the turn count instead, `tick` seconds a turn, - which is what lets a flow declaring six hours be walked to the end of them. Money cannot - be moved at all: a stub runs a model nobody prices, so a dollars cap reads as blind and is - dropped rather than left to never bite. A flow whose only claim is a dollars one is - therefore proved by the turn cap, exactly as one that claimed nothing is. - - Args: - driven: The stubs, which is every agent the flow declared. - declared: What the flow said, or None for a flow with no opinion -- which is left with - no allowance at all, so that the turn cap is what ends it exactly as before. - scenario: The world, for how much clock a turn is worth in it. - steps: The shared turn count, which is what the clock is read off. - - Returns: - The reckoning the stubs are held to, or None where there is nothing to hold them to -- - which is what says whether a `Stopped` was this allowance being reached. - """ - import time - - from hmz.coganchor.agents.allowance import Allowance as Said - from hmz.coganchor.agents.allowance import Ledger as Reckoning - - if declared is None: - return None - # Money dropped: nothing in a proof is priced, so a cap on it could only ever read as - # blind, and a claim that cannot be tried must not be reported as one that failed. - held = Said(hours=declared.hours, tokens=declared.tokens) - if not held.bounded: - return None - began = time.monotonic() - # Patched over the whole child rather than handed to the ledger: it is the child's own - # process, thrown away with the proof, and the ledger reads the clock the same way every - # other reader of it does. A turn is worth `tick` of it, so the clock moves when the flow - # does and stands still while it is not taking turns -- which is what a proof measures. - time.monotonic = lambda: began + steps.taken * scenario.tick - ledger = Reckoning(held, driven) - for agent in driven: - agent.allowance = ledger - return ledger - - -class _Enough(BaseException): - """The turn cap, raised past everything a flow catches: a proof is over when it is. - - A `BaseException`, so that a flow's own `except Exception` -- which is a fine thing for - a loop to write around a turn -- does not swallow the one thing that ends its proof. - """ - - -class _Steps: - """The turns taken so far, shared by every stub of one proof.""" - - def __init__(self, cap: int) -> None: - self.cap = cap - self.taken = 0 - - def step(self) -> None: - """Counts one turn, and ends the proof on the one past the cap.""" - self.taken += 1 - if self.taken > self.cap: - raise _Enough - - -def _driven(flow: str, spec: dict[str, Any]) -> dict[str, Any]: - """Loads one flow and, given a scenario, drives it with stubs to whatever end. - - Args: - flow: The flow, as the parent was given it. - spec: The parent's ask: the name inside the file, the config, and the scenario -- - or None for a proof that only loads. - - Returns: - What the parent folds into the proof: `refused` for a flow that would not load, - `findings` off the live config model, and how driving it went. - """ - import asyncio - import time - - from .driving import NotAFlow, declares, set_up - - # A proof's world sleeps for free. The rest a loop takes between rounds is part of its - # manners and no part of its shape, and it is the shape on trial: a loop that rests - # five seconds a round is not five hundred seconds more legal than one that does not. - # Patched before the flow is even loaded, so a `from time import sleep` reads this one. - time.sleep = _rested - # And the other spelling of it: an async flow rests with `asyncio.sleep`, and one left - # sleeping would be reported as a flow that cannot end when it was only resting. - asyncio.sleep = _rested_for - - named = f"{flow}:{spec['name']}" if spec["name"] else flow - scenario = Scenario(**spec["scenario"]) if spec["scenario"] else None - try: - run, places, make, setting, mark = declares(named) - except NotAFlow as refused: - return {"refused": str(refused)} - except BaseException as raised: # noqa: BLE001 -- reported, in a process built for it - return {"refused": f"the flow's own file raised as it was read -- {raised}"} - answered: dict[str, Any] = {"findings": _styled(setting)} - given = None - if spec["config"] is not None: - try: - given = set_up(named, setting, spec["config"]) - except NotAFlow as refused: - answered["refused"] = str(refused) - return answered - if scenario is None: - return answered - from hmz.coganchor.agents import Stopped - - steps = _Steps(scenario.turns) - settings = () if setting is None else (given,) - held: tuple[dict[str, Any], ...] = ({},) if mark.resumable else () - driven = _crewed(places, scenario, steps) - ledger = _allowed(driven, mark.budget, scenario, steps) - try: - out = run(make(driven), _TASK, *settings, *held) - if inspect.isawaitable(out): - import asyncio - - asyncio.run(_awaited(out)) - finished, turns, said = True, steps.taken, "" - except Stopped as stopped: - # The flow's own declared allowance, run out. It ended, and it ended the way its - # author said it should -- a loop that goes until what it asked for is gone is a loop - # with a bound, and this is that bound being reached rather than a crash. - # - # Asked of the ledger rather than taken from the exception's type: a flow that stops - # its own agent and then takes a turn raises this same `Stopped`, and a flow whose - # loop ends because it stopped itself by mistake is not a flow that finished. - if ledger is not None and ledger.spent: - finished, turns, said = True, steps.taken, "" - else: - finished, turns = False, steps.taken - said = f"stopped without having run out of what it declared -- {stopped}" - except _Enough: - finished, turns = False, scenario.turns - said = ( - f"still going after {scenario.turns} turns -- nothing inside the flow " - "ended it, and a loop is legal when something inside it can end it" - ) - except BaseException: # noqa: BLE001 -- the flow's own crash is the outcome - import traceback - - tail = traceback.format_exc().strip().splitlines() - finished, turns, said = False, steps.taken, "\n".join(tail[-3:]) - answered.update(finished=finished, turns=turns, said=said) - return answered - - -async def _awaited(out: Any) -> None: - """One awaitable flow, awaited: what `asyncio.run` takes is a coroutine.""" - await out - - -def _styled(setting: type[BaseModel] | None) -> list[dict[str, str]]: - """The config findings only the live model can show, said as the static reading says. - - The model a flow declares may be built anywhere -- a helper module, a call -- and the - static reading only checks the ones written in plain sight. This is the same two rules - against the model `declares` actually resolved. - - Args: - setting: The model, or None for a flow that takes no setting up. - - Returns: - One finding per thing found, as plain values for the one JSON line home. - """ - if setting is None: - return [] - found: list[dict[str, str]] = [] - config = setting.model_config - if not (config.get("extra") == "forbid" or config.get("frozen") is True): - found.append( - { - "code": "loose-config", - "severity": "warning", - "said": "the config takes anything -- set model_config to extra: " - "forbid or frozen: True, so a setting that is misspelled is refused " - "rather than quietly ignored", - } - ) - for name, field in setting.model_fields.items(): - if not field.description: - found.append( - { - "code": "unsaid-field", - "severity": "warning", - "said": f"the config field {name!r} says nothing about itself -- " - "give it a Field(description=...), which is what whoever sets the " - "flow up is shown", - } - ) - return found - - -def _crewed( - places: Sequence[Place], scenario: Scenario, steps: _Steps -) -> list[AgentBase]: - """The stub agents for one flow's places, all of one scenario and one turn count. - - Built inside a function because the drivers are heavy and the parent half of this - module is imported by `hmz._legacy_flows` itself: only a child actually proving a flow pays - for them. - - The stubs claim every capability there is -- every moment, a goal feature, shapes, - tools, a turn that takes a word put into it -- because what is being proved is the flow - and not the agents: a flow legal on the widest backend is refused for a narrower one - where the agents are chosen, which is `driving.py`'s job and not this one's. Everything - else is the real base classes, so the hooks a flow hangs fire exactly as they would - under a real backend -- a `Stop` hook that refuses sends a stub on again, and that - continuation is a counted turn. - - Args: - places: What the flow declared. - scenario: The world the stubs answer from. - steps: The shared turn count, whose cap ends a proof nothing else ends. - - Returns: - One agent per place, the person's included. - """ - from hmz.coganchor.agents import ( - AgentBase, - AgentConfig, - Event, - HumanAgent, - Moment, - SessionBase, - Usage, - ) - from hmz.coganchor.agents.human import HumanSession - - class StubSession(SessionBase): - """A turn that lands at once and answers what the scenario says.""" - - shapes: ClassVar[bool] = True - takes_tools: ClassVar[bool] = True - steers: ClassVar[bool] = True - narrates: ClassVar[bool] = True - - def interject(self, text: str) -> None: - # Taken rather than refused, as every capability here is claimed: a stub's turn - # lands the moment it starts, so the word is written down and taken back off in - # one -- a flow that asks whether it may steer is answered the way the backends - # it will really run on answer. - self.took(self.steering(text)) - - def _stream( - self, prompt: str, *, schema: type[BaseModel] | None = None - ) -> Iterator[Event]: - del prompt - steps.step() - if self._id is None: - self._adopt(f"stub-{id(self)}-{steps.taken}") - spent = Usage(output=scenario.climb) - self._spends(spent) - yield Event(kind="result", text=_said(schema, scenario), spent=spent) - - def _pursue(self, objective: str) -> str: - del objective - steps.step() - self._spends(Usage(output=scenario.climb)) - return scenario.answer - - class StubAgent(AgentBase): - """An agent claiming every capability, so only the flow is on trial.""" - - moments: ClassVar[frozenset[Moment]] = frozenset(Moment) - pursues: ClassVar[bool] = True - - def new(self, cwd: str | os.PathLike[str] | None = None) -> StubSession: - return StubSession(self, cwd) - - class StubTalk(HumanSession): - """The person's answers, deterministic: the scenario's, not a prompt's.""" - - shapes: ClassVar[bool] = True - - def stream( - self, prompt: str, *, schema: type[BaseModel] | None = None - ) -> Iterator[Event]: - # Overridden whole, as the real person's session is: their turn is not one - # being watched, and not one bracketed by an agent's moments. It still counts - # against the cap -- a flow that loops on asking forever is a flow that does - # not stop, whoever it is asking. - del prompt - steps.step() - yield Event(kind="result", text=_said(schema, scenario)) - - class StubPerson(HumanAgent): - """The person at the prompt, answering as the scenario has them answer.""" - - def new(self, cwd: str | os.PathLike[str] | None = None) -> StubTalk: - return StubTalk(self, cwd) - - return [ - StubPerson() - if place.person - else StubAgent(AgentConfig(model="stub", effort=""), name=place.name or None) - for place in places - ] - - -def _said(schema: type[BaseModel] | None, scenario: Scenario) -> str: - """What one stub turn answers with, as the text the base classes read back. - - Args: - schema: The shape the turn was held to, or None for a plain turn. - scenario: The world answering. - - Returns: - The scenario's answer for a plain turn; for a shaped one, the fabricated model as - its own JSON -- or "", which the base reads back as no answer at all, for the silent - scenario and for a shape nothing can be fabricated for. - """ - if schema is None: - return scenario.answer - if scenario.verdict is None: - return "" - from pydantic import ValidationError - - try: - return schema.model_validate(_made(schema, scenario)).model_dump_json() - except ValidationError: - return "" - - -def _made(schema: type[BaseModel], scenario: Scenario) -> dict[str, Any]: - """A shaped answer fabricated field by field, deterministically. - - Every boolean says the scenario's verdict -- which is what makes `NEVER_DONE` the - reviewer that never says done, whatever the field is called -- and every string says - its answer. The rest is the quietest legal value: a default where the field has one, - the first of a literal's few, an empty list, a zero, a nested shape made the same way. - - Args: - schema: The shape. - scenario: The world answering. - - Returns: - The fields, ready to be read back through the model. - """ - return { - name: _filled(field.annotation, field, scenario) - for name, field in schema.model_fields.items() - } - - -def _unioned(kind: Any) -> tuple[Any, ...]: - """One annotation and, for a union, what it is a union of. - - Only a union: the arguments of a `list[str]` are what is inside it, not what the field - itself may be, and a list of strings answered as one string would be the confusion. - """ - import types - import typing - - if get_origin(kind) in (types.UnionType, typing.Union): - return (kind, *get_args(kind)) - return (kind,) - - -def _filled(kind: Any, field: Any, scenario: Scenario) -> Any: - """One field's value, off its annotation. - - Args: - kind: What the field was annotated with, unions unwrapped as they are met. - field: The field itself, for the default it may carry. - scenario: The world answering. - - Returns: - The value. - """ - from typing import Annotated - - from pydantic import BaseModel - - if get_origin(kind) is Annotated: - # The constraints ride along in the field itself; what is answered is the type. - return _filled(get_args(kind)[0], field, scenario) - for said in _unioned(kind): - if said is bool: - return scenario.verdict - if said is str: - return scenario.answer - if get_origin(said) is Literal: - return get_args(said)[0] - if field is not None and not field.is_required(): - return field.get_default(call_default_factory=True) - for said in _unioned(kind): - if said in (int, float): - return 0 - if get_origin(said) in (list, tuple, set, frozenset): - # As many as the field says it takes at the least, each made the same way: - # a shape that requires three lanes is answered with three, not refused. - fewest = 0 - for bound in getattr(field, "metadata", None) or (): - fewest = max(fewest, getattr(bound, "min_length", 0) or 0) - inner = next(iter(get_args(said)), None) - return [_filled(inner, None, scenario) for _ in range(fewest)] - if get_origin(said) is dict: - return {} - if isinstance(said, type) and issubclass(said, BaseModel): - return _made(said, scenario) - return None - - -def _main(argv: list[str]) -> None: - """The child's whole life: one flow, one spec, one JSON line back.""" - flow, spec = argv - said = json.dumps({"proving": _driven(flow, json.loads(spec))}) - sys.stdout.write(said + "\n") - sys.stdout.flush() - - -if __name__ == "__main__": - _main(sys.argv[1:]) diff --git a/src/hmz/runtime/flowing/stepping.py b/src/hmz/runtime/flowing/stepping.py deleted file mode 100644 index ea244b41..00000000 --- a/src/hmz/runtime/flowing/stepping.py +++ /dev/null @@ -1,520 +0,0 @@ -"""Running a prophecy: one node at a time, and picking one up where it stopped. - -What an ordinary flow does is whatever its body does, and a run of it that was stopped is a -run that has to start again -- a flow keeps a handful of things in a dict and works out the -rest. An atlas is the other bargain. Its body was compiled before anything ran, so a run of -one is a walk over the prophecy: take the node, run it, write down what it answered, follow -the edge whose guard holds. What a run has done is therefore the list of answers it has, and -picking one up is walking the same prophecy again over the same answers until it reaches the -node that has none. - -Which node that is decides what happens next. By default it runs: a node stopped partway is -work that was not done, and doing it again is the only honest reading of a turn that was cut -off. A node may say otherwise -- `@mind(rerun=False)` -- and is then stepped past, having -already had its effect by the time anything could interrupt it. Such a node answers with -nothing, which is what makes stepping past it possible at all: there is no answer for what -comes next to be missing. - -A run is picked up into the same prophecy or not at all. What was written down is written -down against :func:`~hmz.runtime.flowing.prophecy.digest`, and an atlas rewritten between -two runs of it is a different prophecy whose nodes happen to share their names -- so the -digest is checked, and a run whose prophecy has moved starts from the top rather than -resuming into somewhere it has never been. -""" - -from __future__ import annotations - -from dataclasses import dataclass, field -from pathlib import Path -from typing import TYPE_CHECKING, Any, cast - -from hmz._legacy_flows.atlas import ATLAS - -from .prophecy import AGENTS, CONFIG, INPUT, Node, Reads, digest, shipped - -if TYPE_CHECKING: - import os - from collections.abc import Mapping - - from pydantic import BaseModel - - from hmz._legacy_flows import Agent - - from .driving import Entry - from .prophecy import Edge, Prophecy - -__all__ = ["walking"] - -#: What the state a run writes down keeps: which prophecy it is a run of, which node it was -#: inside when it stopped, and what each node it has finished answered. -_PROPHECY = "prophecy" -_AT = "at" -_DONE = "done" - -#: And that the run reached the way out. What a run has done is kept for whoever reads it -#: back, so a finished one cannot be told from a stopped one by what it holds -- and the next -#: run of the flow, handed that, would walk every answer it already had and do no work at all. -_OVER = "over" - -#: What one visit to a node is written down under: the node, and how many times the run has -#: been through it -- a loop is one node visited again, and a round whose answer overwrote -#: the last round's would be a run that could not be picked up inside a loop. -_VISIT = "#" - -#: And what a supernode's own nodes are written down under: its visit, then theirs. -_UNDER = "/" - - -def walking( - flow: str | os.PathLike[str], inside: Mapping[str, Any], entry: Entry -) -> Entry: - """Compiles one atlas, and answers with something that runs the prophecy. - - Called where a flow is about to be run rather than where it is merely read, which is - what makes an atlas a flow checked before anything happens: a body that does not compile - is a flow refused before its first node, with everything wrong with it said at once. - - Args: - flow: The atlas, as it was asked for -- which is also which of the ones its file holds - is wanted. - inside: What running the flow's file left behind, which is where the nodes are. - entry: The atlas's own entry point, whose body is the declaration that was compiled - and is therefore never called. What it was marked with is carried onto the answer, so - that everything reading a flow off its entry point goes on reading this one. - - Returns: - Something to call the way any flow is called -- the agents, the task, the config for - one that takes one, and the dict a resumable flow is handed. - - Raises: - NotAFlow: If the atlas does not compile, saying each reason on a line of its own. - """ - import functools - - from .driving import NotAFlow - from .finding import inside as which - from .finding import reading - from .prophesying import named_as, prophesied - - named = str(flow) - under = Path(reading(named)) - wanted = named_as(under, which(named)) - prophecy = _shipped(under, wanted) - if prophecy is None: - held = prophesied(under, name=which(named)) - if held.prophecy is None: - why = "\n".join( - f" {one.where}:{one.line}: {one.code}: {one.said}" - for one in held.findings - if one.severity == "error" - ) - raise NotAFlow(f"{flow}: the atlas does not compile\n{why}") - prophecy = held.prophecy - walked = prophecy - - def running(agents: Any, task: Any, *said: Any) -> Any: - # A resumable flow is handed its state last and its config before it, and an atlas - # is always resumable: what a run of one has done is which of its nodes answered. - state: dict[str, Any] = said[-1] if said else {} - config = said[0] if len(said) > 1 else None - return _stepped( - walked, inside, agents, task, _set_up(walked, inside, config), state - ) - - # Whatever the entry point was marked with, and not the two marks known today: both - # `flow` and `atlas` set theirs into the function's own `__dict__`, which is exactly - # what this copies -- so a third mark added later travels without this line moving. - return functools.update_wrapper(running, entry, assigned=(), updated=("__dict__",)) - - -def _set_up( - prophecy: Prophecy, inside: Mapping[str, Any], config: BaseModel | None -) -> BaseModel | None: - """What a run of an atlas is set up with, which is its defaults where nobody said. - - An atlas's body has no way to make one: `config or Config()` is work, and work is what a - node is for. So the model's own defaults stand in, and the compiling refuses a config - that cannot be built out of them -- which is what makes this always answer with one. - - Args: - prophecy: The compiled atlas, for the model it says it can be set up with. - inside: What running its file left behind, which is where that model is. - config: What the run was set up with, or None for one nobody set up. - - Returns: - The config, and None for an atlas that takes no setting up. - """ - from pydantic import BaseModel - - if config is not None or not prophecy.config: - return config - model = inside.get(prophecy.config) - if not (isinstance(model, type) and issubclass(model, BaseModel)): - return None - try: - return model() - except Exception: # noqa: BLE001 -- a model the compiling said could be built and now - return None # cannot is a file rewritten under the run, not a run to end here - - -def _shipped(under: Path, wanted: str) -> Prophecy | None: - """The prophecy a flow's own directory ships, where it ships the one being asked for. - - Preferred over compiling the atlas again: the compiling is where an atlas is refused, - and a repository that has been through it has an answer worth carrying. A directory - holds one prophecy and a file may hold several atlases, so the one shipped is the one it - is named after -- and the rest are compiled where they are asked for. - - Args: - under: The flow's own directory, or the file a single-file flow is, which ships none. - wanted: Which of the atlases the file holds is being run. - - Returns: - The prophecy, or None where the flow ships none, or ships one for another atlas. - - Raises: - NotAFlow: If it ships one that cannot be read back. Refused rather than compiled - again: what a flowverse shipped is what it meant to be run, and quietly running - something else would be the one thing shipping it was meant to rule out. - """ - from .driving import NotAFlow - - held = shipped(under) - if held is None: - return None - if held.prophecy is None: - raise NotAFlow( - f"{held.at}: the prophecy shipped here cannot be read -- compile the atlas " - "again, or take the file away and let the run compile it" - ) - return held.prophecy if held.prophecy.name == wanted else None - - -@dataclass(slots=True) -class _Walk: - """One prophecy being walked, and everything every step of it is against. - - Held once rather than handed down: what changes as a run goes is which node it is at and - what each name holds, and everything else is the same at every step. A supernode makes - another of these -- its own prophecy, its own file, its own agents -- and keeps what the - whole run shares. - - Attributes: - prophecy: The compiled atlas being walked. - inside: What running the file it was compiled from left behind, which is where the - functions its nodes are get looked up. - agents: The run's agents, by the name this prophecy calls each. - state: The whole run's state, which is where it says what node it stopped inside. - kept: What each visit to a node that has answered answered, for the whole run. - under: What this prophecy's visits are written down beneath, "" for the outermost. - beside: What each flow reached by name left behind when it was read, for the whole - run. Read once: a run walks one prophecy, and a sub-flow re-read between two rounds - of a loop would be new code running under a graph that had already been settled. - nodes: The prophecy's nodes by id, and its ways out by the node each leaves. Built - once rather than scanned at every step, a prophecy being what it is for the length - of the walk. - ways: As above. - """ - - prophecy: Prophecy - inside: Mapping[str, Any] - agents: dict[str, Agent] - state: dict[str, Any] - kept: dict[str, Any] - under: str - beside: dict[str, Mapping[str, Any]] - nodes: dict[str, Node] = field(init=False) - ways: dict[str, tuple[Edge, ...]] = field(init=False) - - def __post_init__(self) -> None: - """Reads the prophecy into what a step of the walk asks it.""" - self.nodes = {one.at: one for one in self.prophecy.nodes} - self.ways = { - at: self.prophecy.out_of(at) - for at in {"", *(one.out_of for one in self.prophecy.edges)} - } - - -def _stepped( - prophecy: Prophecy, - inside: Mapping[str, Any], - agents: Any, - given: Any, - config: BaseModel | None, - state: dict[str, Any], -) -> Any: - """Runs one whole atlas, from whatever a run of it has already done. - - Args: - prophecy: The compiled atlas. - inside: What running its file left behind. - agents: The agents, as the atlas declared them. - given: What the atlas was called with -- the task, or the shape a supernode takes. - config: What the run was set up with, or None for an atlas that takes no setting up. - state: What the run before this one wrote down, and what this one writes into. - - Returns: - Whatever the prophecy answers with, and None for one that answers with nothing. - """ - written = digest(prophecy) - if state.get(_PROPHECY) != written or state.get(_OVER): - # A different prophecy: the atlas was rewritten between the two runs, so what the - # last one did, it did somewhere else. Cleared rather than merged, since a node - # that kept its name is not thereby the node it was. And the same for a run that - # reached the way out: it is a run to read back rather than one to pick up, and - # picking it up would be a run with an answer for every node and nothing to do. - state.clear() - state[_PROPHECY] = written - walk = _Walk( - prophecy=prophecy, - inside=inside, - agents={one: getattr(agents, one) for one in prophecy.agents}, - state=state, - kept=state.setdefault(_DONE, {}), - under="", - beside={}, - ) - answered = _walked(walk, given, config) - # Written down as finished rather than emptied: what a run did is what whoever reads it - # back is after, and the next run of the flow is what must not be handed it. - state[_OVER] = True - _saved(state) - return answered - - -def _walked(walk: _Walk, given: Any, config: BaseModel | None) -> Any: - """Walks one prophecy from its way in to its way out. - - Args: - walk: The prophecy being walked, and what every step of it is against. - given: What it was called with. - config: What the run was set up with. - - Returns: - What the prophecy answers with, which is what the `return` it left by named. - """ - bound: dict[str, Any] = {AGENTS: walk.agents, INPUT: given, CONFIG: config} - seen: dict[str, int] = {} - at = "" - while True: - edge = _way(walk, at, bound) - node = None if edge is None else walk.nodes.get(edge.into) - if node is None: - # What the `return` named, and not whatever the last node happened to say: a - # body may answer with something it bound three nodes ago. - return bound.get(edge.answers) if edge is not None else None - seen[node.at] = visit = seen.get(node.at, 0) + 1 - held = f"{walk.under}{node.at}{_VISIT}{visit}" - answered = _answered(walk, bound, node, held) - if node.binds: - bound[node.binds] = answered - at = node.at - - -def _way(walk: _Walk, at: str, bound: dict[str, Any]) -> Edge | None: - """Which way out of one node this run takes. - - Args: - walk: The prophecy being walked. - at: The node it is leaving, and "" for the way in. - bound: What each name holds now. - - Returns: - The edge. One whose far end is "" is the way out of the prophecy, and what the run - answers with is on it; None is a node nothing leads on from, which nothing that - compiled can be. - """ - for edge in walk.ways.get(at, ()): - when = edge.when - if ( - when is None - or bool(_read(bound, Reads(when.reads, when.field))) is when.truth - ): - return edge - return None - - -def _answered(walk: _Walk, bound: dict[str, Any], node: Node, held: str) -> Any: - """What one node answers with: what it answered last time, or what it answers now. - - Args: - walk: The prophecy being walked. - bound: What each name holds now. - node: The node. - held: What this visit to it is written down under. - - Returns: - Its answer, rebuilt through the shape it declared where this is a visit the run is - picking up rather than one it is taking. - """ - if held in walk.kept: - return _rebuilt(walk.kept[held], node.gives, walk.inside) - if held == walk.state.get(_AT) and not node.rerun: - # Where the last run stopped, in a node that says it is not to be run again: it had - # its effect before anything could interrupt it, so the run steps past. It answers - # with nothing -- the compiling refuses one that does not -- so there is nothing for - # what comes next to be missing. - walk.kept[held] = None - _saved(walk.state) - return None - # Written down before the node runs and saved once: `State` saves itself as it is - # written into, and what goes into `kept` below is a change inside a value it holds and - # cannot see -- which is the one that has to ask. Where a run stopped is the node that - # was running, so a supernode writes nothing here: the nodes under it write themselves, - # and one of theirs overwritten by this would be a run picked up past what it stopped in. - if node.kind != "atlas": - walk.state[_AT] = held - answered = _ran(walk, bound, node, held) - walk.kept[held] = _written(answered) - _saved(walk.state) - return answered - - -def _ran(walk: _Walk, bound: dict[str, Any], node: Node, held: str) -> Any: - """Runs one node for real: a turn, a Python function, or a whole prophecy. - - Args: - walk: The prophecy being walked. - bound: What each name holds now. - node: The node. - held: What this visit to it is written down under. - - Returns: - What it answered. - - Raises: - NotAFlow: If the file the prophecy was compiled from no longer holds what it declared, - which is a flow rewritten under a run of it. - """ - from .driving import NotAFlow - - said = [_read(bound, one) for one in node.takes] - if node.kind == "atlas": - return _supernode(walk, node, held, said) - call = walk.inside.get(node.calls) - if not callable(call): - raise NotAFlow( - f"{walk.prophecy.name}: nothing in the flow is called {node.calls!r} -- the " - "prophecy was compiled from a file that has since been rewritten" - ) - return call(*said) - - -def _supernode(walk: _Walk, node: Node, held: str, said: list[Any]) -> Any: - """Runs one supernode, which is a whole prophecy inside this one. - - Args: - walk: The prophecy the node is in. - node: The node. - held: What this visit to it is written down under, which its own nodes go beneath. - said: What it was handed -- the agents, then the one shape it takes. - - Returns: - What the prophecy under it answered with. - - Raises: - NotAFlow: If the prophecy names a supernode it does not hold, which nothing that - compiled should be able to say. - """ - from .driving import NotAFlow - from .finding import find, loaded - - under = walk.prophecy.under(node.under) - if under is None: - raise NotAFlow( - f"{walk.prophecy.name}: nothing under it is called {node.under!r}" - ) - # Beside it or elsewhere: a supernode of this flow's own is in the file this prophecy - # was compiled from, and one reached by name is a flow of its own, run to be read as any - # flow is. Told apart by the mark rather than by the name being one this file holds: - # `inner = sub("inner")` binds that name here too, and a membership test would read a - # flow of its own as one beside it and walk its nodes against the wrong file. Read once - # for the run rather than once a visit: the graph was settled before the run started, - # and a file re-read between two rounds of a loop would be new code running under a - # shape that had already been agreed. - if getattr(walk.inside.get(node.calls), ATLAS, None) is not None: - beside = walk.inside - elif node.calls not in walk.beside: - walk.beside[node.calls] = beside = loaded(find(node.calls)) - else: - beside = walk.beside[node.calls] - return _walked( - _Walk( - prophecy=under, - inside=beside, - agents={one: walk.agents[one] for one in under.agents}, - state=walk.state, - kept=walk.kept, - under=f"{held}{_UNDER}", - beside=walk.beside, - ), - said[1], - None, - ) - - -def _read(bound: dict[str, Any], one: Reads) -> Any: - """What one node reads, which is a name a node bound or a field of it. - - Args: - bound: What each name holds now. - one: The reading. - - Returns: - The value. Nothing at all for a name nothing has bound, which the compiling refuses - and which a file rewritten under a run could still produce. - """ - held = bound.get(one.reads) - if not one.field: - return held - # The agents are the one thing a name holds that is not a node's answer, and they are - # held by name: everything else a body binds is a model or a plain kind. - if isinstance(held, dict): - return cast("dict[str, Any]", held).get(one.field) - return getattr(held, one.field, None) - - -def _written(answered: Any) -> Any: - """One node's answer as something a run picked up again can be handed back. - - Args: - answered: What the node answered. - - Returns: - JSON for a model, and the value itself for the plain kinds -- which is the whole of - what a node may answer with, the compiling having refused everything else. - """ - dump = getattr(answered, "model_dump", None) - return dump(mode="json") if callable(dump) else answered - - -def _rebuilt(written: Any, gives: str, inside: Mapping[str, Any]) -> Any: - """One node's kept answer, back in the shape the node declared. - - Args: - written: What was written down. - gives: The shape the node answers with, and "" for one that answers with nothing. - inside: What running the flow's file left behind, which is where the model is. - - Returns: - The value, rebuilt through the model where the flow still declares one -- and as it - was written where it does not, a run picked up into a rewritten flow being one whose - prophecy has already been checked against the digest. - """ - validate = getattr(inside.get(gives), "model_validate", None) - return validate(written) if callable(validate) and written is not None else written - - -def _saved(state: dict[str, Any]) -> None: - """Writes the run's state where it is kept, for a run stopped after this node. - - A `hmz.runtime.epic.State` saves itself as it is written into, and a plain dict is what a flow - run from a test is handed. Both are dicts here, and only one of them has anything to do: - what is written inside `done` is written inside a value the mapping cannot see change. - - Args: - state: What the run is writing down. - """ - save = getattr(state, "save", None) - if callable(save): - save() diff --git a/src/hmz/runtime/kept.py b/src/hmz/runtime/kept.py index a12f6906..60f247a1 100644 --- a/src/hmz/runtime/kept.py +++ b/src/hmz/runtime/kept.py @@ -1,126 +1,70 @@ """What an agent is, written down. -An agent is a CLI, an account, a model at an effort, and the machine its work lands on. What -it may do, the goals it may reach for, whether it searches the web and the skills it carries -are not its own to hold: the first three are the flow's, declared where the flow declares the -agent's place, and the last are the CLI's, installed and switched off where that CLI keeps -them. So this is a shape and the two directions it goes in, and nothing else. +An agent is a CLI, an account, and a model at an effort. What it may do, the capabilities it +is granted and the skills it carries are not its own to hold: they are the flow's, declared +where the flow declares the agent's role. Where its work lands is not its own either: that is +the environment the flow opens its session in. So this is a shape and the two directions it +goes in, and nothing else -- the same word `-a` takes after the role. Here rather than beside the interface because both ways in read it: the interface writes an agent down per workspace per flow and a command line reads the same file back, and a command line that had to load a terminal interface to read a file of six lines would be paying for a layer it does not use. - -How one agent goes into a file is here rather than beside the settings that hold it for the -reason it always was: an agent is written the same way wherever it is written, and two places -writing one shape is two places to drift. """ from __future__ import annotations -from typing import Any, NamedTuple +from typing import NamedTuple __all__ = ["Runs", "read_back", "written"] class Runs(NamedTuple): - """What one agent of a flow was set up to run, and where its turns land. + """What one agent of a flow was set up to run. Attributes: - spec: The agent itself, as `cli/model:effort` -- the same word a command line takes. - anchor: The machine its work lands on, as a target, or "" to work on this one. - permission: What it may do without being asked, as one of `hmz.coganchor.agents.PERMISSIONS`, - or "" for nothing said about it, which leaves the agent at what it was configured - with. An answer withheld rather than a rung named quietly on somebody's behalf. + spec: The agent itself, as `cli/model:effort`. provider: The account its turns run as, by the name a provider of its CLI was made under, or "" to run as this machine is already signed in. - goals: Whether backend goals are available. This is always an on/off answer; any - suggestion attached to the flow's agent place is resolved before this is constructed. - web_search: Whether it may search the web, or None for nothing said about it. Three - states rather than the two `goals` has, and the reason the two of them part company - here is what happens to the answer afterwards: this one is settled onto the agent's - config, and a `True` written down for an agent nobody asked about is an answer that - switches searching on wherever the CLI had it off. The old two-state reading of this - field -- on is what an agent nobody was asked about does -- was that answer, given - every time this was built and never noticed because it agreed with the default it was - overwriting. So the silence gets a value of its own and stays a silence all the way - to the command line, and off and on go on being written down as the answers they are. """ spec: str - anchor: str = "" - permission: str = "" provider: str = "" - goals: bool = True - web_search: bool | None = None - -def written(runs: Runs) -> dict[str, Any]: - """One agent as it goes into a file, which is the shape a workspace's settings hold. - Read from both ends, as a command line reads one: a model may hold slashes of its own, - while a CLI and an effort never do. +def written(runs: Runs) -> str: + """One agent as it goes into a file: the word `-a` takes after `=`. Args: runs: What the agent is. Returns: - Its fields, less any that says nothing -- an agent that works here, may do whatever an - agent nobody was asked about may do, and runs as this machine is signed in is one every - field of which is the field's own silence. + `cli@provider/model:effort`, or `cli/model:effort` for one that runs as this machine + is signed in. """ cli, _, rest = runs.spec.partition("/") - model, _, effort = rest.rpartition(":") - held: dict[str, Any] = {"cli": cli, "model": model, "effort": effort} - if runs.anchor: - held["anchor"] = runs.anchor - if runs.permission: - held["permission"] = runs.permission - if runs.provider: - held["provider"] = runs.provider - # Both values are material: on may be an override of a workflow whose default is off, so - # what is written down always records the explicit two-way choice. Web search is written - # the same way and for the same reason -- and its silence is written too, as the `null` - # it is. Left out, it could not be told from a file written before there was such a - # setting, and those are the one case that has to read as on: that is what every agent - # did then. A key that is there and empty is this file saying nobody was asked. - held["goals"] = runs.goals - held["web_search"] = runs.web_search - return held - - -def read_back(held: dict[str, Any], *, goals: bool = True) -> Runs | None: + account = f"@{runs.provider}" if runs.provider else "" + return f"{cli}{account}/{rest}" + + +def read_back(said: object) -> Runs | None: """One agent as it comes back off a file, or None where what is there is not one. + Read from both ends, as a command line reads one: a model may hold slashes and colons of + its own, while a CLI and an effort never do. + Args: - held: What the file holds for it. - goals: Whether goals are available where the entry does not say -- which is every entry - written before there was such a setting. + said: What the file holds for it. Returns: - The agent, or None for an entry written by hand and not the way these are written. + The agent, or None for an entry written by hand, or by an older humanize, and not the + way these are written. """ - cli, model, effort = held.get("cli"), held.get("model"), held.get("effort") - if not (cli and model and effort): + if not isinstance(said, str): + return None + head, slash, rest = said.partition("/") + cli, _, provider = head.partition("@") + model, colon, effort = rest.rpartition(":") + if not (slash and colon and cli and model and effort): return None - # An entry that says nothing about what it may do runs at what an agent nobody has been - # asked about has always run at; one that names no account runs as this machine is signed - # in. A `skills` an older file holds is the CLI's own business now, and is read past. - said = held.get("goals") - # Asked for with a default rather than read off what `get` returns for a key that is not - # there, because here the two are different answers: a file holding `null` is one that - # was asked and says nobody answered, and a file holding nothing at all is older than the - # question. - searches = held.get("web_search", True) - return Runs( - f"{cli}/{model}:{effort}", - str(held.get("anchor") or ""), - str(held.get("permission") or ""), - str(held.get("provider") or ""), - said if isinstance(said, bool) else goals, - # An entry written before there was such a setting is one whose agent searched the - # web, that being what every agent did then. A key holding anything that is not an - # answer -- the `null` this writes, or something somebody typed in by hand -- holds - # nothing anybody may act on, and it stays nothing. - searches if isinstance(searches, bool) else None, - ) + return Runs(f"{cli}/{model}:{effort}", provider) diff --git a/src/hmz/runtime/runner.py b/src/hmz/runtime/runner.py index d7d655db..0c2c891a 100644 --- a/src/hmz/runtime/runner.py +++ b/src/hmz/runtime/runner.py @@ -1,663 +1,828 @@ -"""What starts a flow: the file it is in, the agents it takes, and the line naming both. +"""What starts a flow: the line naming it, the drivers it is handed, and the run written down. The line is read here rather than beside the command that carries it out, because the terminal -interface starts a flow from that same line and then keeps the agents -- which is what lets -something typed while the flow runs reach the one working. A reader that lived in the command -line would be one the interface had to reach up into. - -What a flow is, and what it says it drives, is :mod:`hmz._legacy_flows`. This asks it, hands the -flow -the agents it declared under the names it calls them, and writes the run down as an epic. -Nothing a flow itself reaches for is here: a flow names one module of humanize's, and it is -not this one. +interface starts a flow from the same parts: a flow, what each of its agent and environment +roles is given, its params, and what the run may spend. A reader that lived in the command line +would be one the interface had to reach up into. + +A run is four steps, and the first three refuse before anything runs. The flow is loaded and +what it declares is read (:mod:`hmz.runtime.flowing.finding`); what each role is given is +checked against that declaration and opened as a driver +(:func:`~hmz.runtime.flowing.harnesses.open_agent`, +:func:`~hmz.runtime.flowing.environments.open_env`) -- which starts no CLI and reaches no +machine; each environment given is probed, which does reach it; and then the flow is run by +:func:`~hmz.runtime.flowing.engine.run_flow`, written down as it goes into an epic that also +holds the engine's journal of a flow that can be picked up. What refuses a run before it has +started is :class:`Refused`, which a command line reports as a line to correct. + +What a flow is, and everything it is handed, is :mod:`hmz.flows` and the engine behind it. +Nothing a flow itself reaches for is here: a flow names one module of humanize's, and it is not +this one. """ from __future__ import annotations +import asyncio +import contextlib +import json +import math from pathlib import Path -from typing import TYPE_CHECKING, Any, cast - -from hmz.coganchor import backends +from typing import TYPE_CHECKING, Any, NamedTuple, cast if TYPE_CHECKING: import os - from argparse import ArgumentParser - from collections.abc import Awaitable, Mapping, Sequence + from collections.abc import Callable, Iterable, Mapping + + from hmz.coganchor.agents import AgentBase, SessionBase + from hmz.flows import Budget, FlowParams, Usage + from hmz.runtime.flowing import ( + AgentDriver, + Declaration, + EnvDriver, + FlowImpl, + LiveCall, + OutworlderDriver, + SessionHandle, + ) + from hmz.runtime.flowing.harnesses import Listener + from hmz.runtime.flowing.specs import AgentSpec, EnvSpec + + from .epic import Drove, Epic + +__all__ = ["Line", "Refused", "Runner", "read_line"] + +#: What is told of each session a run opens, as it is opened: the role it was opened for, and +#: the coganchor agent and conversation behind it. +type Opened = Callable[[str, AgentBase, SessionBase], None] + - from pydantic import BaseModel +class Refused(ValueError): # noqa: N818 -- named for what happened, as the flow API names its own + """A run refused before anything of it ran: a line, or a setup, to correct. - from hmz.coganchor.agents import AgentBase - from hmz.coganchor.agents.allowance import Allowance - from hmz.runtime.flowing.driving import Entry + The message says what, in words; the exception it was refused for is its cause. + """ + + +class Line(NamedTuple): + """An `hmz exec` line, read. + + Attributes: + flow: The flow, as the line named it. + task: What it is to do. + agents: What each agent role is given, in the order the line wrote them. + envs: What each environment role is given, likewise. + params: Each param as the line wrote it, which the flow's own model reads. + budget: What the run may spend, or None where the line said nothing. + resume: Whether to pick up the newest run of the flow here that can be. + as_json: Whether a program is reading the run rather than a person. + """ -__all__ = ["Runner", "flow_and_agents", "read_agent", "set_up_from"] + flow: str + task: str + agents: tuple[AgentSpec, ...] = () + envs: tuple[EnvSpec, ...] = () + params: dict[str, str] = {} # noqa: RUF012 -- a NamedTuple's default, never written to + budget: Budget | None = None + resume: bool = False + as_json: bool = False -def _finished(running: Awaitable[None]) -> None: - """Runs a flow that is a coroutine, until it returns. +def read_line(argv: list[str]) -> Line: + """Reads an `hmz exec` line. - A flow may be written as ``async def run``, which is how one drives many agents at once: - the loop is the flow's own, started here and closed when the flow returns, so that a flow - which awaits nothing and one which awaits ten thousand turns are both just run. Starting - the flow is the same call either way -- whatever is driving one is driving a flow, not an - event loop, and none of them has to know which kind it took. + Only the line: which roles the flow has, and whether it needs a budget, are asked of the + flow by :class:`Runner`, so that `--help` loads no flow and pays for no driver. Args: - running: The flow, as the coroutine calling it made. + argv: What followed the command name. + + Returns: + The line. + + Raises: + SystemExit: For a line that is not one, as argparse rejects it -- an unknown flag, no + flow or task, or an `-a`, `-e`, `-p` or `-b` that cannot be read. """ - import asyncio - import contextvars - from concurrent.futures import ThreadPoolExecutor + import argparse - async def flowing() -> None: - # A coroutine of our own around it: `asyncio.run` takes one of those, and what a - # flow answered with is whatever awaiting it is spelled as where the flow was written. - await running + from hmz.coganchor import backends + + parser = argparse.ArgumentParser( + prog="hmz exec", description="Run an agent flow in this directory." + ) + parser.add_argument( + "-f", + "--flow", + required=True, + metavar="FLOW", + help="the flow to run: one humanize ships or a flowverse holds, by name, a directory " + "or file of your own, or a git+URL#flow ref; `:` for another flow of " + "the same module", + ) + parser.add_argument( + "-a", + "--agents", + action="append", + default=[], + metavar="ROLE=SPEC[,...]", + help="what an agent role runs: ROLE=CLI[@PROVIDER]/MODEL:EFFORT, several to one " + "option separated by commas, the option repeated as often as suits. CLI is one of " + f"{', '.join(sorted(one.name for one in backends.profiles()))}", + ) + parser.add_argument( + "-e", + "--envs", + action="append", + default=[], + metavar="ROLE=SPEC[,...]", + help="where an environment role is: ROLE=local@/abs/path or " + "ROLE=ssh@[user@]host[:port]/abs/path (ssh@host/~/path under the login's home). " + "A role the runtime fills -- the workspace -- is never named", + ) + parser.add_argument( + "-p", + "--params", + action="append", + default=[], + metavar="KEY=VALUE[,...]", + help="a param of the flow; a value is read as the param's type, or as JSON", + ) + parser.add_argument( + "-b", + "--budget", + action="append", + default=[], + metavar="KEY=VALUE[,...]", + help="what the run may spend: duration=1h30m, cost=5 (USD), output_tokens=200k, " + "graceful=false to stop a turn mid-way. Required, except for chat", + ) + parser.add_argument( + "--resume", + action="store_true", + help="pick up the newest run of this flow here, for a flow that can be picked up", + ) + parser.add_argument( + "--json", + action="store_true", + dest="as_json", + help="write the run as NDJSON on stdout -- one object per thing an agent says, " + "flushed as it is said -- for a program to read instead of a person", + ) + parser.add_argument( + "task", + help="what the flow is to do, after -- if it starts with a dash", + ) + args = parser.parse_args(argv) + + from hmz.runtime.flowing.specs import ( + SpecError, + parse_agents, + parse_budget, + parse_envs, + parse_params, + ) try: - asyncio.get_running_loop() - except RuntimeError: - asyncio.run(flowing()) # nothing is turning here, which is the ordinary way in - return - # Started from a thread that is already running a loop of its own -- an interface, a test. - # A flow cannot be run on that one: it would be the flow waiting for turns that are - # waiting for the loop the flow is holding, which is a run that never takes its first. - # - # The context goes with it, since a thread is otherwise handed an empty one: what the run - # was entered as is held there, and a flow that called another from a thread that had - # never heard of the run would be a flow with no branch to be on. - with ThreadPoolExecutor(max_workers=1, thread_name_prefix="humanize-flow") as apart: - apart.submit(contextvars.copy_context().run, asyncio.run, flowing()).result() + return Line( + flow=args.flow, + task=args.task, + agents=tuple(parse_agents(args.agents)), + envs=tuple(parse_envs(args.envs)), + params=parse_params(args.params), + budget=parse_budget(args.budget) if args.budget else None, + resume=args.resume, + as_json=args.as_json, + ) + except SpecError as bad: + parser.error(str(bad)) class Runner: - """A flow, loaded from a file and handed the agents it was written for. - - A flow is a Python file with a ``run(agents: tuple[...], task: str)`` in it, and the tuple - is how many agents it drives -- the one thing about a flow that cannot be read off the - command line starting it. Checking it before anything runs is what keeps a two-agent flow - started with one agent from failing on an unpacking hours into a loop, with a turn's work - already behind it. A flow that declares a NamedTuple instead has also said what each of - its agents is for, and they are called that from here on. + """A flow, loaded, with a driver for every role it was given and everything checked. + + Nothing has run once one is made, and nothing has been started: an agent's CLI is reached + when its first session opens, and an environment's machine when it is probed, which + :meth:`arun` does before the flow is called. """ def __init__( self, flow: str | os.PathLike[str], - agents: Sequence[AgentBase], - config: BaseModel | dict[str, Any] | None = None, - resume: str | os.PathLike[str] | None = None, - container: str = "", - budget: Allowance | Mapping[str, Any] | None = None, + *, + agents: Mapping[str, str | AgentDriver] | Iterable[AgentSpec] = (), + envs: Mapping[str, str | EnvDriver] | Iterable[EnvSpec] = (), + params: Mapping[str, Any] | FlowParams | None = None, + budget: Budget | Mapping[str, Any] | None = None, + resume: bool | str | os.PathLike[str] = False, + workspace: str | os.PathLike[str] | None = None, ) -> None: - """Loads the flow and holds the agents to drive it with. + """Loads the flow and checks what it is given against what it declares. Args: - flow: The Python file the flow is written in. It is run to be read, so whatever it - does as it is imported happens here, and fails here as it would anywhere. - agents: The agents to hand it, as many as it declares. - config: What it was set up with, for a flow that says it can be -- an instance of - the model :func:`configures` answers with, or the fields to build one from, which - is what a YAML file of them reads as. None is a flow left as it comes, and is - what a flow that takes no setting up is given either way. - resume: The epic to pick up from, for a flow that says it can be picked up: the - state that run left behind is what this one is handed. None is the last run of - this flow here, which is what running a resumable flow again means -- a loop - meant to run for a week is one that carries on where it stopped. A flow that - says nothing about being resumable ignores this, having nowhere to put it. - container: The image to run the whole of this in, or "" to run it on this machine. - A convenience rather than a second way of saying where an agent works: it starts - one container, points every agent of the run at it, and lets the flow's own code - reach it through `hmz._legacy_flows.container()`, which is the name a flow writes for - what `hmz.runtime.flowing.driving` holds -- and which is what a run in a - container is, said once from outside rather than agent by agent inside. - budget: What this run may spend, as an `Allowance` or the three fields to build one - from -- which is what a `budget:` in a YAML file reads as. None takes the flow's - own default, and the flow having none is a run under nothing at all. Given - rather than read off the flow, because what a run is worth is whoever started it - to say and the flow only ever said a default. + flow: The flow, as `-f` names one. + agents: What each agent role runs: an `-a` spec after `=` or a driver, by + role, or the specs a line read. A role the runtime fills -- an `Outworlder` -- + is never given. + envs: What each environment role is: an `-e` spec after `=` or a driver, by + role, or the specs a line read. A `LocalEnv` role is never given; it is the + workspace. + params: The flow's params, as its model or as a mapping of values -- strings from + `-p` among them -- or None for its defaults. + budget: What the run may spend. Only a flow humanize ships may be run without one, + under `Budget(cost=inf)`. + resume: Whether to pick up the newest run of this flow in the workspace that can + be picked up, or the epic to pick up. + workspace: Where the run happens, defaulting to this directory. Raises: - NotAFlow: If the flow is not there, is not a flow -- nothing in it marked - ``@flow()``, or one whose ``agents`` cannot be read or says nothing about how many - it takes -- or is a - flow that drives a different number of agents than were given, or one of them - cannot run a moment the flow said that place has to, or was set up with something - that is not what it asked for, or brings a skill from a repository that cannot be - reached. + Refused: For a flow that cannot be loaded; a role given that it does not declare, + that the runtime fills, or that is given twice; a required role left out; an + agent that is not the harness its role names or cannot do what its role asks; a + spec a driver cannot be made for; params the flow does not take; no budget; and a + run to pick up that is not there or of a flow that cannot be picked up. """ - from hmz.coganchor.agents import HumanAgent - from hmz.coganchor.agents.allowance import allowed - from hmz.runtime.flowing.driving import ( - NotAFlow, - carries, - declares, - lands, - readies, - runs_at, - serves, - set_up, - ) + from hmz.flows import Budget, FlowException + from hmz.runtime.flowing import builtin, resolved - from .epic import resumed - - run, places, make, setting, mark = declares(flow) - # Before anything is chosen or opened: an atlas whose body does not compile is a - # flow refused where the run is set up rather than from inside one that has already - # pulled an image and opened an epic. - readies(run) - if config is not None: - config = set_up(flow, setting, config) - asked = [place for place in places if not place.person] - if len(asked) != len(agents): - raise NotAFlow( - f"{flow}: the flow drives {len(asked)} agents, {len(agents)} given" + self._named = str(flow) + self._workspace = Path(workspace) if workspace is not None else Path.cwd() + try: + impl = resolved(self._named) + declared = impl.describe() + except FlowException as why: + raise Refused(str(why)) from why + self._impl: FlowImpl = impl + self._declared = declared + agents_given, self._specs = self._agents_of(agents) + envs_given, self._places = self._envs_of(envs) + try: + self._params = impl.params_of({} if params is None else params) + except FlowException as why: + raise Refused(str(why)) from why + if budget is None: + if not builtin(impl): + raise Refused( + f"{self._named}: a run is given a budget -- -b duration=...,cost=...," + "output_tokens=... -- and this one was given none" + ) + budget = Budget(cost=math.inf) + self._budget = budget if isinstance(budget, Budget) else _budget(budget) + self._picked_up = self._picks_up(resume) + # Made last, once everything that could refuse the run has had its say: a driver + # starts nothing as it is made, and none is made for a run that is refused. + self._agents = _agent_drivers(agents_given) + self._envs = _env_drivers(envs_given) + self._recorder: Recorder | None = None + + # ------------------------------------------------------------------ what is checked + + def _agents_of( + self, given: Mapping[str, str | AgentDriver] | Iterable[AgentSpec] + ) -> tuple[dict[str, AgentSpec | AgentDriver], dict[str, str]]: + """What each agent role was given, checked, and the spec each was written as. + + Raises: + Refused: For a role that cannot be given this. + """ + from hmz.runtime.flowing.specs import AgentSpec, SpecError, parse_agents + from hmz.runtime.flowing.spi import HARNESS_CAPABILITIES + + named = self._named + drivers: dict[str, AgentSpec | AgentDriver] = {} + specs: dict[str, str] = {} + for name, given_as in _by_role(given): + role = self._declared.agent(name) + if role is None: + raise Refused( + f"{named} has no agent role {name!r}; its agent roles are " + f"{_roles(one.name for one in self._declared.agents if not one.auto)}" + ) + if role.auto: + raise Refused( + f"{named}: {name!r} is filled by the runtime -- whoever is outside " + "the run -- and is not given with -a" + ) + if name in drivers: + raise Refused(f"{named}: the agent role {name!r} is given twice") + try: + said = ( + parse_agents([f"{name}={given_as}"])[0] + if isinstance(given_as, str) + else given_as + if isinstance(given_as, AgentSpec) + else cast("AgentDriver", given_as) + ) + except SpecError as why: + raise Refused(str(why)) from why + if isinstance(said, AgentSpec): + harness, capabilities = said.harness, HARNESS_CAPABILITIES[said.harness] + else: + harness, capabilities = said.harness, said.capabilities + if role.harness is not None and harness != role.harness: + raise Refused( + f"{named}: {name!r} is {role.harness}, and {harness} was given" + ) + if lacking := role.capabilities - capabilities: + raise Refused( + f"{named}: {name!r} needs " + f"{', '.join(sorted(one.__name__ for one in lacking))}, which " + f"{harness} does not do" + ) + drivers[name] = said + if isinstance(said, AgentSpec): + specs[name] = str(said).partition("=")[2] + else: + account = f"@{said.provider}" if said.provider else "" + specs[name] = ( + f"{said.harness}{account}/{said.model}:{said.effort or 'auto'}" + ) + if missing := [ + one.name + for one in self._declared.agents + if one.required and not one.auto and one.name not in drivers + ]: + raise Refused( + f"{named} needs an agent for {_roles(missing)}; give each with " + "-a ROLE=CLI/MODEL:EFFORT" ) - # Whatever a place declared that its agent's backend had no way of carrying, for a - # place that said it would rather run than be refused. Kept rather than printed: this - # is the same shape as `unreadable`, where the object that knows answers and whoever - # has a screen does the saying. - unserved: list[str] = [] - # Before the first turn, for the reason the count is: a flow that hangs a hook on a - # moment its agent does not run would otherwise find out hours into a loop, from a - # hook that raised where it was hung rather than from the line that chose the agent. - for agent, place in zip(agents, asked, strict=True): - if short := place.moments - type(agent).moments: - raise NotAFlow( - f"{flow}: {place.name or 'the agent'} has to run " - f"{', '.join(sorted(short))}, which {agent.backend} does not" + return drivers, specs + + def _envs_of( + self, given: Mapping[str, str | EnvDriver] | Iterable[EnvSpec] + ) -> tuple[dict[str, EnvSpec | EnvDriver], dict[str, str]]: + """What each environment role was given, checked, and the spec each was written as. + + Raises: + Refused: For a role that cannot be given this. + """ + from hmz.runtime.flowing.specs import EnvSpec, SpecError, parse_envs + + named = self._named + drivers: dict[str, EnvSpec | EnvDriver] = {} + specs: dict[str, str] = {} + for name, given_as in _by_role(given): + role = self._declared.env(name) + if role is None: + raise Refused( + f"{named} has no environment role {name!r}; its environment roles " + f"are {_roles(one.name for one in self._declared.envs if not one.auto)}" ) - if place.goal and not type(agent).pursues: - raise NotAFlow( - f"{flow}: {place.name or 'the agent'} is run under a goal, which " - f"{agent.backend} has no feature for" + if role.auto: + raise Refused( + f"{named}: {name!r} is the workspace the run is started in, and is " + "not given with -e" ) - if place.goal and not agent.goals_enabled: - raise NotAFlow( - f"{flow}: {place.name or 'the agent'} is run under a goal, but goals " - "were switched off for it" + if name in drivers: + raise Refused(f"{named}: the environment role {name!r} is given twice") + try: + said = ( + parse_envs([f"{name}={given_as}"])[0] + if isinstance(given_as, str) + else given_as + if isinstance(given_as, EnvSpec) + else cast("EnvDriver", given_as) ) - serves(flow, agent, place) - # Told what the whole run will be put in, because nothing is pointed at that - # container until the run starts: a place needing somewhere remote must not be - # refused here and then allowed when a flow called another inside the same run. - lands(flow, agent, place, container=container) - # And what the flow says this one may do, whether it has goals and whether it - # reads the internet -- over whatever it was made with, because those three are - # the flow's and nobody else's: whoever chose the agent chose a CLI, a model, an - # effort and an account, and none of that says what the work is. - runs_at(flow, agent, place, dropped=unserved) - # The person at the prompt is made here rather than given: nobody chooses what they - # run, so nothing upstream of this was ever asked about them. - given = iter(agents) - driven = [HumanAgent() if place.person else next(given) for place in places] - for agent, place in zip(driven, places, strict=True): - if place.name: - agent.rename(place.name) - # What the flow works by, mounted onto every session these agents open. Before the - # first turn, since a repository the flow named is fetched to get it: a run that - # cannot reach one says so here rather than an hour into a loop. - carries(flow, driven) - self._unserved = tuple(unserved) - self._run: Entry = run - # Only for a flow that said it takes one, so that every flow written before there - # was such a thing is still called with the two arguments it declares. - self._config: BaseModel | None = config if setting is not None else None - self._setting = setting - # The drivers themselves, which is what the run is written down out of and what - # whoever started the flow reaches for: the person the flow talks to is among them, - # having been made here rather than chosen. - self._driven = tuple(driven) - # And the same agents as the flow declared them: a flow whose agents are a NamedTuple - # reaches them by name, and one that unpacks a plain tuple sees no difference. - self._agents = make(driven) - self._flow = str( - flow - ) # as it was named, which is what a run of it is named after - #: Whether the flow says it can be picked up where the last run of it left off, and - #: which run that was. Asked here rather than when the run starts, so that an epic - #: named at the prompt is one whoever named it hears about before anything runs. - self._resumable = mark.resumable - #: What this run may spend, settled here so that every way of starting a flow reaches - #: one answer: what the line or the menu said, else what the flow declared, else - #: nothing at all. What the flow declared is kept beside it, because the two together - #: are what says whether an unbounded run is one anybody meant. - self._declared = mark.budget - self._budget = allowed(budget, mark.budget) - #: The image the whole run works in, or "" for a run on this machine. The container - #: is started as the flow starts rather than here: constructing a runner reads a - #: flow, and reading one must not pull an image. - self._container = container - self._picked_up: Path | None = None - if self._resumable: - self._picked_up = ( - Path(resume) if resume is not None else resumed(self._flow) + except SpecError as why: + raise Refused(str(why)) from why + drivers[name] = said + if isinstance(said, EnvSpec): + specs[name] = str(said).partition("=")[2] + else: + # As `-e` spells one, which a run picked up is given again. + workdir = str(said.workdir).lstrip("/") + specs[name] = f"{said.backend}@{said.provider}/{workdir}" + if missing := [ + one.name + for one in self._declared.envs + if one.required and not one.auto and one.name not in drivers + ]: + raise Refused( + f"{named} needs an environment for {_roles(missing)}; give each with " + "-e ROLE=BACKEND@PROVIDER/WORKDIR" ) + return drivers, specs - @property - def agents(self) -> tuple[AgentBase, ...]: - """Every agent this drives, in the order the flow takes them. + def _picks_up(self, resume: bool | str | os.PathLike[str]) -> Path | None: # noqa: FBT001 + """The epic this run picks up, or None for a run from the top. - Which is not what it was given: a flow that says it talks to the person is driving - one more agent than anybody chose, and whatever is driving the flow has to reach - that one too -- it is the one thing here that answers with what was typed. + Raises: + Refused: For a flow that cannot be picked up, or no run of it to pick up. """ - return self._driven + from .epic import picks_up, resumed + + if resume is False: + return None + if not self._impl.resumable: + raise Refused( + f"{self._named} does not say it can be picked up, so there is no run of " + "it to resume" + ) + if resume is True: + found = resumed(self._impl.ref, self._workspace) + if found is None: + raise Refused( + f"{self._named} has no run here to pick up: none got as far as " + "writing anything down" + ) + return found + found = Path(resume) + if not picks_up(found): + raise Refused(f"{found.name} holds nothing a run could be picked up from") + return found + + # ----------------------------------------------------------------------- what it is @property - def budget(self) -> Allowance: - """What this run will be held to, whoever or whatever settled it.""" - return self._budget + def flow(self) -> str: + """The flow, as it was named.""" + return self._named @property - def unwatched(self) -> bool: - """Whether nothing at all will stop this run and nobody has said that is the point. + def impl(self) -> FlowImpl: + """The flow, loaded.""" + return self._impl - Asked of a runner rather than worked out again wherever one is started, so that the - menu's second confirmation and the command line's line on stderr are the same - question about the same run. + @property + def declaration(self) -> Declaration: + """What the flow declares.""" + return self._declared - A cap this run's agents cannot read is handed in as no cap: a dollars cap on a model - nobody prices is a run with nothing to stop it, whatever the file it was written in - says, and one that said so in a log line and nowhere else was one nobody was asked - about. - """ - from hmz.coganchor.agents.allowance import unwatched + @property + def agents(self) -> dict[str, AgentDriver]: + """The driver each agent role was given, by role.""" + return dict(self._agents) - return unwatched(self._budget, self._declared, self._blind()) + @property + def envs(self) -> dict[str, EnvDriver]: + """The driver each environment role was given, by role.""" + return dict(self._envs) - def unreadable(self) -> str: - """Which of the caps this run was given nothing in it can read, in words. + @property + def params(self) -> FlowParams: + """The flow's params, validated.""" + return self._params - Answered before the first turn rather than at the end of a run that never stopped: a - dollars cap on a model nobody prices is a cap that cannot bite, and it reads exactly - like one that has not bitten yet. + @property + def budget(self) -> Budget: + """What the run may spend.""" + return self._budget - Returns: - One line about them, or "" where every cap set can be read. - """ - from hmz.coganchor.agents.allowance import unreadable + @property + def picked_up(self) -> Path | None: + """The epic this run picks up, or None for a run from the top.""" + return self._picked_up - return unreadable(self._blind()) + @property + def workspace(self) -> Path: + """Where the run happens.""" + return self._workspace - def _blind(self) -> frozenset[str]: - """Which caps this run was given nothing driving it can read. + @property + def recorder(self) -> Recorder | None: + """What is writing the run down, once it has started, or None before.""" + return self._recorder - Read off the agents rather than off the allowance, and read here rather than in each - of the two things that ask: whether a cap can be read at all is a fact about what this - run drives, so a run answered `unwatched` and a run answered `unreadable` are answered - about the same agents. + def unreadable(self) -> str: + """Which cap of the run nothing it drives can read, in words, or "" for none. - Returns: - The dimensions, as `Reading.blind` names them. + A cost cap over an agent whose model nobody prices is a cap that cannot bite: its + turns cost nothing anybody can count, which reads exactly like a cap that has not bitten + yet. Answered before the first turn rather than at the end of a run that never stopped. """ - from hmz.coganchor.agents.allowance import Ledger + from hmz.coganchor.prices import price - return Ledger(self._budget, self._driven).reads().blind + cost = self._budget.cost + if cost is None or math.isinf(cost): + return "" + unpriced = sorted( + {one.model for one in self._agents.values() if price(one.model) is None} + ) + if not unpriced: + return "" + return ( + f"nobody lists a price for {', '.join(unpriced)}, so cost={cost:g} cannot " + "stop what it spends" + ) - def unserved(self) -> str: - """Which of this flow's declarations its agents' backends could not carry, in words. + def watch(self, listener: Listener) -> None: + """Has everything every session of the run says reach `listener`. - A place that wrote `insist=False` would rather run on a backend that cannot be told - than not run at all, and what it gets is the declaration dropped rather than applied - quietly. Dropped is the honest half; said is the other half, and this is where the - saying starts -- answered on the runner, which is what settled them, and worded by - whoever has a screen: `hmz exec` on stderr, the interface in the transcript. + What a way in shows a run through: every session is watched for what it spends, + which also stops a CLI writing its own progress to this process's streams. - Returns: - One line per declaration given up, or "" for a run that carried everything its - flow declared -- which is every run of every flow that did not ask for leniency. + Args: + listener: What to tell, from whichever thread a CLI is read on. """ - return "\n".join(self._unserved) + for driver in self._agents.values(): + watch = getattr(driver, "watch", None) + if callable(watch): + watch(listener) - def run(self, task: str) -> None: - """Runs the flow in this directory, for as long as it keeps running. + # ------------------------------------------------------------------------- running - The run is written down as it happens: which agents were driven, at what, and which - sessions each of them opened. Nothing else knows a session was part of a run -- the - backends log them one by one, under ids of their own -- and the run is over the moment - this returns, however it returns. - - A flow written as ``async def run`` is run to its return here too, on a loop of its - own: this waits for the flow either way, so that whatever started one is holding a - run rather than a coroutine somebody has to remember to await. + async def arun( + self, + task: str, + *, + outworlder: OutworlderDriver | None = None, + opened: Opened | None = None, + started: Callable[[Epic], None] | None = None, + ) -> Any: + """Runs the flow to its return, on the loop this is awaited on. Args: - task: What the flow is to have its agents do. - """ - import inspect + task: What it is to do. + outworlder: Whoever is outside the run, or None for nobody -- an outworlder that is + always away, which is what a command line is. + opened: What is told of each session as it opens, or None. + started: What is handed the epic the run is written into, once it is open. - from hmz.coganchor.agents.allowance import Ledger - from hmz.runtime.flowing.driving import contained, entered, lands_in, left + Returns: + What the flow returned. + + Raises: + Refused: If an environment cannot be reached, or the drivers do not meet what the + flow declares -- before the flow has been called. + BaseException: Whatever the flow raised, as it raised it. + """ + from hmz.flows import ( + BudgetExceeded, + FlowCancelled, + FlowDefinitionError, + FlowException, + ParamsError, + RequirementError, + ) + from hmz.runtime.flowing import local_env, open_outworlder, probe, run_flow - from .epic import Epic, state + from .epic import Epic from .settings import Settings - # Written down as running before it is: what a flow calls is written down the same - # way, so that whatever is watching reads one list of what is running under what, - # rather than a flow it was told about and a flow it was not. - started = entered(self._flow, self._driven) - picked_up = self._picked_up try: - # One container for the run, started here rather than where the runner was made: - # reading a flow must not pull an image, and a run that never starts must not - # leave one behind. Every agent is pointed at it as it comes up, and what the - # flow itself reads, writes and runs there is `hmz._legacy_flows.container`. - with ( - contained(self._container) as where_, - Epic( - self._flow, - self._driven, - task, - resumable=self._resumable, - picked_up=picked_up.name if picked_up is not None else "", - # Whether this workspace asked for its runs to be profiled as well as - # traced, which is a thing about the project being worked on: a repository - # whose tests take an hour is a different question from one whose take a - # minute. Read here rather than in the epic, which is the run written down - # rather than the settings under it. - profile=Settings().profiling, - ) as epic, - ): - # One reckoning for the run, and every agent holds it: an allowance is the - # run's money rather than any one agent's, and a clone made mid-flow joins - # it as it is made. Here rather than wherever a run is started from, so that - # `hmz exec`, the interface and a flow calling another all get it -- nothing - # a flow can be started by has to remember to hang one on. - ledger = Ledger(self._budget, self._driven) - for agent in self._driven: - agent.epic = epic - agent.allowance = ledger - if where_ is not None: - lands_in(self._driven, where_) - # As it was set up, or as it comes: a flow that takes a config takes None - # for the run nobody set up, which is the default the flow declared. And - # after it, for a flow that says it can be picked up, what the run it is - # being picked up from left behind -- which is a dict it writes into, kept - # in this run's own epic as it writes. - said: list[Any] = [self._agents, task] - if self._setting is not None: - said.append(self._config) - if self._resumable: - said.append( - epic.state( - self._flow, - state(picked_up, self._flow) - if picked_up is not None - else None, - ) + for driver in self._envs.values(): + await probe(driver) + local = local_env(self._workspace) + except BaseException as why: + # Stopped, or refused, before the run began: what it was given goes either way. + await asyncio.shield(self._closed(None)) + if isinstance(why, FlowException): + raise Refused(str(why)) from why + raise + impl = self._impl + epic = Epic( + self._named, + task, + self._workspace, + ref=impl.ref, + agents=[_drove(role, spec) for role, spec in self._specs.items()], + envs=[f"{role}={spec}" for role, spec in self._places.items()], + # Through JSON text rather than `mode="json"`, which leaves an infinite cost a + # float: `Budget(cost=inf)` is written as the string it reads back from. + params=json.loads(self._params.model_dump_json()), + budget=json.loads(self._budget.model_dump_json()), + resumable=impl.resumable, + picked_up=self._picked_up, + profile=Settings(self._workspace).profiling, + ) + recorder = Recorder(epic, opened) + self._recorder = recorder + try: + with epic: + if started is not None: + started(epic) + try: + return await run_flow( + impl, + task, + agents=self._agents, + envs=self._envs, + params=self._params, + budget=self._budget, + outworlder=outworlder or open_outworlder(), + journal=epic.resume if impl.resumable else None, + resume=self._picked_up is not None, + local=local, + recorder=recorder, + ) + except (RequirementError, ParamsError, FlowDefinitionError) as why: + # Refused by the engine before the flow was called: the drivers do not + # meet what it declares, or a skill a role names is not to be had. + if not recorder.started: + raise Refused(str(why)) from why + raise + except (asyncio.CancelledError, FlowCancelled, BudgetExceeded): + epic.stopped() + raise + finally: + usage = recorder.usage() + epic.write( + "usage", + cost=usage.cost, + output_tokens=usage.output_tokens, + seconds=usage.duration.total_seconds(), ) - running_now = self._run(*said) - # Read off what the call answered rather than off the function: a flow is what - # it does when it is called, and one wrapped in something of its own -- a - # decorator that times its rounds -- is the same flow. - if inspect.isawaitable(running_now): - _finished(running_now) finally: - left(started) + await asyncio.shield(self._closed(local)) + async def aclose(self) -> None: + """Closes every driver the run was given, for a run that will not be run after all.""" + await self._closed(None) -def read_agent(spec: str) -> tuple[str, backends.Profile, str, str, str]: - """Reads and validates one command-line agent specification. + async def _closed(self, local: EnvDriver | None) -> None: + """Closes every driver the run was given, and the workspace's, however it ended.""" + for driver in (*self._agents.values(), *self._envs.values(), local): + if driver is None: + continue + with contextlib.suppress(Exception): + await driver.close() - The grammar itself is `hmz.coganchor.backends.read`, an agent being a backend before it is - anything else. This is the name the line's own reading of one goes by, kept because that is what - the spec calls it -- and holding nothing of its own, since everything it used to check moved - into the grammar when the written-out spelling went. + def run(self, task: str, *, outworlder: OutworlderDriver | None = None) -> Any: + """Runs the flow to its return, on a loop of its own in this thread. - Args: - spec: One agent, as `-a` spells one. An `-a` naming several is split into them first. + Args: + task: What it is to do. + outworlder: Whoever is outside the run, or None for nobody. - Returns: - The place the agent fills -- "" for one the line left to fill a place in order -- the - backend, model, effort and provider. What the agent may do, whether it has goals and - whether it may search the web are not among them: those are the flow's, said where it - declares the place, and a line that says one is a line to correct. + Returns: + What the flow returned. + """ + return asyncio.run(self.arun(task, outworlder=outworlder)) - Raises: - ValueError: If the specification is malformed, or says what the flow says. - """ - return backends.read(spec) +class Recorder: + """What writes a run down as the engine runs it: a record per flow call, and each session. -def flow_and_agents( - argv: list[str], -) -> tuple[str, list[AgentBase], str, dict[str, Any] | None, Allowance | None, bool]: - """Reads an `hmz exec` line into a flow, the agents to drive it, the task, and its setup. + Answers to :class:`hmz.runtime.flowing.engine.Recorder`. Every call is told on the loop + the run is on. - A flow says how many agents it drives and what it calls each of them, and this is where - they come from: one for each, at the model and effort each is to run at, in the order the - flow takes them or each naming the place it fills. + Attributes: + started: Whether the flow the run was started with has been called, which is what + tells a run refused before it started from one that failed. + """ - Args: - argv: What followed the command name. + def __init__(self, epic: Epic, opened: Opened | None = None) -> None: + """Holds the epic to write into, and what to tell of each session opened.""" + self._epic = epic + self._opened = opened + self._records: dict[int, Epic] = {} + self._sessions: list[SessionHandle] = [] + self.started = False + + def entered(self, call: LiveCall) -> None: + """A flow call started: the run's own, or one written into a record of its own.""" + self.started = True + if call.parent is None: + self._records[id(call)] = self._epic + return + above = self._records.get(id(call.parent), self._epic) + self._records[id(call)] = above.called(call.ref) + + def left(self, call: LiveCall, error: BaseException | None) -> None: + """A flow call ended, which closes its record.""" + from hmz.flows import BudgetExceeded, FlowCancelled + + from .epic import Sub + + record = self._records.pop(id(call), None) + if not isinstance(record, Sub): + return + stopped = isinstance( + error, asyncio.CancelledError | FlowCancelled | BudgetExceeded + ) + record.ended( + None if error is None else type(error), "stopped" if stopped else "" + ) - Returns: - The flow's path, the agents to drive it with in the order the flow takes them, the task, - what to set the flow up with -- the YAML file `-c` named, read but not yet checked - against the flow's own model, or None where the line named none -- what that file said - the run may spend or None where it said nothing, and whether a program is reading the - run rather than a person. + def spawned( + self, call: LiveCall, role: str, session: SessionHandle, driver: AgentDriver + ) -> None: + """A flow call opened a session: named for its role, and written into its record.""" + self._sessions.append(session) + record = self._records.get(id(call), self._epic) + agent: AgentBase | None = getattr(session, "agent", None) + if agent is None: + # A driver with no coganchor agent behind it -- a fake -- names its session as + # it opens it, and is written down then. + record.session( + role, str(driver.harness), driver.provider, session.id or "?" + ) + return + # Named for its role, and written down in the record of the call that opened it + # once its CLI has said what it calls the conversation, which is its first turn. + agent.rename(role) + agent.epic = record + conversation: SessionBase | None = getattr(session, "coganchor", None) + if self._opened is not None and conversation is not None: + self._opened(role, agent, conversation) - Raises: - SystemExit: If the line does not name a flow and an agent apiece, names a place the flow - has not got, or names a config that cannot be read, as argparse rejects it. - """ - import argparse + @property + def sessions(self) -> tuple[SessionHandle, ...]: + """Every session the run has opened, oldest first.""" + return tuple(self._sessions) + + def usage(self) -> Usage: + """Everything the run's sessions have spent, up to the moment it is read.""" + import datetime + + from hmz.flows import Usage + + cost = 0.0 + tokens = 0 + seconds = 0.0 + for session in tuple(self._sessions): + said = session.usage + cost += said.cost + tokens += said.output_tokens + seconds += said.duration.total_seconds() + return Usage( + duration=datetime.timedelta(seconds=seconds), + cost=cost, + output_tokens=tokens, + ) - parser = argparse.ArgumentParser( - prog="hmz exec", description="Run an agent flow in this directory." - ) - parser.add_argument( - "-f", - "--flow", - required=True, - metavar="FLOW", - help="the flow to drive: one humanize ships or a flowverse holds, by name, or a file " - "of your own; `:` for one of several in a file", - ) - parser.add_argument( - "-a", - "--agent", - action="append", - # One agent for each the flow drives, which for a flow that talks only to the person - # at the prompt is none: the person is handed over rather than chosen, so a line that - # named one would be naming what nobody picks. A line short of an agent the flow does - # need is caught where every other miscount is, by the flow's own declaration. - default=[], - dest="agents", - metavar="SPEC[,SPEC...]", - help="the agents to drive the flow with, each [NAME=]CLI[@PROVIDER]/MODEL:EFFORT -- " - "several to one option, separated by commas, and the option repeated as often as " - "suits. Unnamed they fill the flow's places in the order it takes them; NAME fills " - "the place the flow calls that, and either every one of them names a place or none " - f"does. CLI is one of {', '.join(sorted(one.name for one in backends.profiles()))}", - ) - parser.add_argument( - "-c", - "--config", - metavar="PATH", - help="a YAML file of what to set the flow up with, one field per line, as the flow " - "declares them; only for a flow that says it can be set up. Its `budget:` is the " - "run's own rather than the flow's -- `{hours: 6, tokens: 10, dollars: 50}`, each " - "0 or absent for no limit on that one", - ) - parser.add_argument( - "--json", - action="store_true", - dest="as_json", - help="write the run as NDJSON on stdout -- one object per thing an agent says, " - "flushed as it is said -- for a program to read instead of a person", - ) - parser.add_argument( - "task", - help="what the flow is to have the agents do, after -- if it starts with a dash", - ) - args = parser.parse_args(argv) - held: dict[str, Any] | None = None - budget: Allowance | None = None - if args.config is not None: - try: - held, budget = set_up_from(args.config) - except ValueError as why: - parser.error(str(why)) - - # Only now that the line is known to name agents: `--help` has already exited, and it - # should not have paid for three backends to say what it takes. - from hmz.coganchor.agents import driver - from hmz.coganchor.agents.base import identifying - - agents: list[AgentBase] = [] - places: list[str] = [] - # The list is split here rather than where one agent is read: every `-a` on the line adds - # to the same list, so what the line names is one list however it was typed -- and one - # mistyped agent among three is then reported as itself rather than as all three. - for spec in (one for said in args.agents for one in said.split(",")): - try: - place, profile, model, effort, provider = read_agent(spec) - except ValueError as bad: - parser.error(f"bad agent {spec!r}: {bad}") - agent, config = driver(profile.name) - try: - # What it may do, whether it has goals and whether it may search the web are - # left as they come: `Runner` settles all three from what the flow declared, - # which is the one place any of them is said. - configured = config( - model=model, - effort=effort, - provider=provider, - # And which CLI it is, where the class does not say so by itself: a line - # naming a CLI somebody added by hand is driven by the one class that drives - # all of them, and the name on the line is the only thing that tells it - # which. `identifying` is where the same argument is written out. - **identifying(config, profile.name), - ) - agents.append(agent(configured)) - except ValueError as bad: - parser.error(f"bad agent {spec!r}: {bad}") - places.append(place) - return ( - args.flow, - _as_declared(parser, args.flow, agents, places), - args.task, - held, - budget, - args.as_json, - ) +def _agent_drivers( + given: Mapping[str, AgentSpec | AgentDriver], +) -> dict[str, AgentDriver]: + """A driver per agent role, made for each role given a spec. -def _as_declared( - parser: ArgumentParser, - flow: str, - agents: list[AgentBase], - places: list[str], -) -> list[AgentBase]: - """Puts the agents in the order the flow takes them, for a line that named their places. + Raises: + Refused: For a spec its CLI cannot be configured at. + """ + from hmz.flows import FlowException + from hmz.runtime.flowing.harnesses import open_agent + from hmz.runtime.flowing.specs import AgentSpec - A line that named none is in that order already, having been written in it. One that named - them is read against what the flow declares here, before anything runs: an actor handed - the reviewer's place is an hour of the wrong work, and which places there are is a - question the flow answers without being given any agents at all. + try: + return { + role: open_agent(said) if isinstance(said, AgentSpec) else said + for role, said in given.items() + } + except FlowException as why: + raise Refused(str(why)) from why - Args: - parser: The line, for reporting one to correct. - flow: The flow, as the line named it. - agents: The agents, in the order the line named them. - places: What each was named for, "" for one the line named no place for. - Returns: - The same agents, in the order the flow takes them. +def _env_drivers(given: Mapping[str, EnvSpec | EnvDriver]) -> dict[str, EnvDriver]: + """A driver per environment role, made for each role given a spec. Raises: - SystemExit: If some of them name a place and some do not, if the flow calls its agents - nothing, or if the names are not one apiece of the ones it declares. + Refused: For a spec whose machine is known not to have its workdir. """ - if not any(places): - return agents - if not all(places): - parser.error( - "name every agent or none of them: an agent that names no place fills the flow's " - "next one, which cannot be counted while the others are filled by name" - ) - from hmz.runtime.flowing.driving import NotAFlow, drives + from hmz.flows import FlowException + from hmz.runtime.flowing.environments import open_env + from hmz.runtime.flowing.specs import EnvSpec try: - declared = drives(flow) - except NotAFlow: - # A flow that cannot be read is `Runner`'s to report and not this line's: reading one - # here is for the names, and a line refused twice is refused in two voices. - return agents - if not declared: - # A flow that has nobody to choose for it is a line with one agent too many, which - # is a miscount like every other and `Runner`'s to report. - return agents - if not any(declared): - parser.error( - f"{flow} declares a plain tuple and calls the agents it drives nothing, so they " - "are given in the order it takes them rather than by name" - ) - for place in places: - if place not in declared: - parser.error( - f"{flow} drives no agent called {place}; it drives {', '.join(declared)}" - ) - if places.count(place) > 1: - parser.error( - f"{flow} drives one agent called {place}, and the line names " - f"{places.count(place)}" - ) - if unfilled := [one for one in declared if one not in places]: - parser.error( - f"{flow} also drives {', '.join(unfilled)}, which the line names nothing for" - ) - held = dict(zip(places, agents, strict=True)) - return [held[one] for one in declared] + return { + role: open_env(said) if isinstance(said, EnvSpec) else said + for role, said in given.items() + } + except FlowException as why: + raise Refused(str(why)) from why -def set_up_from( - said: str | os.PathLike[str], -) -> tuple[dict[str, Any] | None, Allowance | None]: - """Reads what a flow is to be set up with, and what the run may spend, out of a file. +def _by_role(given: object) -> list[tuple[str, object]]: + """What each role was given, whether by role or as the specs a line read.""" + from collections.abc import Mapping - The file is what the flow menu would have asked, written down: one field per - line, under the names the flow declared. It is not checked here -- the flow's own model - is what checks it, and the model is not there until the flow is loaded. + if isinstance(given, Mapping): + return [ + (str(role), said) + for role, said in cast("Mapping[object, object]", given).items() + ] + return [ + (str(getattr(one, "role", "")), one) for one in cast("Iterable[object]", given) + ] - One key of it is reserved and is not the flow's: `budget`, which is the run's allowance - rather than a setting of the flow. Taken out here rather than left in, because the flow's - own model refuses a field it never declared -- and it is a mapping of the three - dimensions rather than a bare number, which is refused loudly for saying nothing about - which of the three it meant. - Args: - said: The path to the YAML. +def _drove(role: str, spec: str) -> Drove: + """One agent role as the epic writes it down, off the spec it was given as.""" + from .epic import Drove + from .kept import read_back - Returns: - What it holds field by field with the reserved key taken out, or None where it left - nothing for the flow at all -- an empty file, or one that says only what the run may - spend, which is not a flow set up with nothing but a flow left as it comes. And the - run's allowance, or None where the file said nothing about one. + runs = read_back(spec) + cli, _, rest = (runs.spec if runs is not None else spec).partition("/") + model, _, effort = rest.rpartition(":") + return Drove(role, cli, model, effort, runs.provider if runs is not None else "") + + +def _roles(names: Iterable[str]) -> str: + """Some roles, as a line says them.""" + said = [repr(one) for one in names] + return ", ".join(said) if said else "none" + + +def _budget(said: Mapping[str, Any]) -> Budget: + """A budget written down as JSON, read back. Raises: - ValueError: If the file cannot be read, holds something that is not a mapping, or says - a budget that cannot be read as one. + Refused: For one that is not a budget. """ - import yaml + import pydantic - from hmz.coganchor.agents.allowance import KEY, written + from hmz.flows import Budget try: - held = yaml.safe_load(Path(said).read_text(encoding="utf-8")) - except (OSError, UnicodeDecodeError, yaml.YAMLError) as why: - raise ValueError(f"cannot read {said}: {why}") from why - if held is None: - return None, None - if not isinstance(held, dict): - raise ValueError( # noqa: TRY004 -- a file to correct, not a caller's type error - f"{said}: a flow is set up from a mapping, not a {type(held).__name__}" - ) - fields = cast("dict[str, Any]", held) - # Copied rather than popped in place: what was handed in is the caller's, and a reader - # that emptied it would be a file that reads differently the second time it is read. - rest = {name: value for name, value in fields.items() if name != KEY} - return rest or None, written(fields[KEY], str(said)) if KEY in fields else None + return Budget.model_validate(dict(said)) + except pydantic.ValidationError as why: + raise Refused(f"the budget is not one: {why}") from why diff --git a/src/hmz/runtime/settings.py b/src/hmz/runtime/settings.py index 7ad8f5b6..2d2a8349 100644 --- a/src/hmz/runtime/settings.py +++ b/src/hmz/runtime/settings.py @@ -1,9 +1,10 @@ """What humanize remembers: what was set up to run here, and what is true everywhere. One file under humanize's own home. Most of it is one entry per workspace -- the flow that was -last run there, and for each flow the workspace has run, what each of its agents was running -- -so a project driven by one flow on two agents is driven by them again tomorrow, rather than -falling back to the default every time it is opened. Beside those is the handful of settings +last run there, and for each flow the workspace has run, what each of its roles was given, what +it was set up with and what a run of it may spend -- so a project driven by one flow on two +agents is driven by them again tomorrow, rather than falling back to the default every time it +is opened. Beside those is the handful of settings that are not a workspace's at all, which is what `enable_sentry` is: whether humanize reports its own failures, answered once and true wherever it is run from. @@ -13,10 +14,9 @@ menu wrote. Kept per flow rather than per workspace alone, because what an agent runs is only meaningful -against the flow that drives it: a flow's second agent is its reviewer, and the flow before it -had no second agent at all. And keyed by what the flow calls each one where it calls them -anything, so that a flow which grows an agent in the middle does not silently hand the -reviewer's model to the builder. +against the flow that drives it: a flow's `reviewer` is its own, and the flow before it had no +reviewer at all. And keyed by the role each fills, so that a flow which grows a role in the +middle does not silently hand the reviewer's model to the builder. There is nowhere else an agent is written down. What an agent is -- a CLI, an account, a model at an effort -- is short enough now that a template kept under a name was more to hold in step @@ -37,7 +37,7 @@ from hmz.runtime.kept import Runs, read_back, written if TYPE_CHECKING: - from collections.abc import Sequence + from collections.abc import Mapping __all__ = ["Settings"] @@ -130,42 +130,40 @@ def forget(self, workspace: str = "") -> bool: self._write() return True - def agents( - self, flow: str, goal_defaults: Sequence[bool] | None = None - ) -> list[Runs]: - """What each agent of one flow was last running here, and where its turns landed. + def agents(self, flow: str) -> dict[str, Runs]: + """What each agent role of one flow was last given here. Args: flow: The flow they were driving. - goal_defaults: What each agent place currently suggests. Used only for an entry - written before goal selection was stored; with none, goals default on. Returns: - One `cli/model:effort` apiece with the machine it was anchored to, what it may do - without being asked and the account it ran as, in the order the flow takes them, - and nothing at all for a flow this workspace has not run. + One agent per role, in the order they were written down, and nothing at all for a + flow this workspace has not run -- or one whose entry this did not write, which + reads as nothing remembered rather than as half of something. """ - flows: dict[str, Any] = self._mine().get("flows") or {} - kept: dict[str, Any] = flows.get(flow) or {} - agents: dict[str, Any] = kept.get("agents") or {} - said: list[Runs] = [] - for at, raw in enumerate(agents.values()): - if not isinstance(raw, dict): - return [] # written by hand and not the way this writes it - # An anchor is what a workspace that has one has: an entry written before there - # were any is a workspace whose agents work here, which is what leaving it out - # already meant. - runs = read_back( - cast("dict[str, Any]", raw), - goals=goal_defaults[at] - if goal_defaults is not None and at < len(goal_defaults) - else True, - ) + said: dict[str, Runs] = {} + for role, raw in self._kept(flow, "agents").items(): + runs = read_back(raw) if runs is None: - return [] - said.append(runs) + return {} + said[role] = runs return said + def envs(self, flow: str) -> dict[str, str]: + """What each environment role of one flow was last given here, as `-e` spells one. + + Args: + flow: The flow they were for. + + Returns: + One `@/` per role, and nothing at all for a flow that + was given none here -- one whose environments are the workspace it runs in. + """ + held = self._kept(flow, "envs") + if not all(isinstance(one, str) for one in held.values()): + return {} + return {role: str(one) for role, one in held.items()} + def flows(self) -> dict[str, Any]: """What every flow this workspace has run was last set up with, by flow. @@ -180,12 +178,12 @@ def flows(self) -> dict[str, Any]: held = self._mine().get("flows") return cast("dict[str, Any]", held) if isinstance(held, dict) else {} - def config(self, flow: str) -> dict[str, Any]: - """How one flow was last set up here, for a flow that can be set up at all. + def params(self, flow: str) -> dict[str, Any]: + """How one flow was last set up here, as its params. - Kept beside what its agents run and for the same reason: a flow of forty settings is - not one to answer again every morning. Read back through the flow's own model rather - than trusted, so a setting the flow has since dropped or renamed is one the model + Kept beside what its roles were given and for the same reason: a flow of forty params + is not one to answer again every morning. Read back through the flow's own model + rather than trusted, so a param the flow has since dropped or renamed is one the model refuses rather than one that quietly comes back. Args: @@ -195,21 +193,20 @@ def config(self, flow: str) -> dict[str, Any]: What was set, field by field, and nothing at all for a flow this workspace has never set up. """ - return self._kept(flow, "config") + return self._kept(flow, "params") def budget(self, flow: str) -> dict[str, Any]: """What a run of one flow here was last said to be allowed to spend. Beside what the flow was set up with rather than inside it, because it is not one of - the flow's settings: the flow said at most a default, and this is what the person - running it here decided a run of it is worth. + the flow's params: it is what the person running it here decided a run of it is worth. Args: flow: The flow it was set for. Returns: - The dimensions that were set, and nothing at all for a flow nobody has set one for - here -- which is a run under whatever the flow itself says. + The budget as JSON -- `duration`, `cost`, `output_tokens`, `graceful` -- and + nothing at all for a flow nobody has set one for here. """ return self._kept(flow, "budget") @@ -224,48 +221,43 @@ def _kept(self, flow: str, under: str) -> dict[str, Any]: What was written down, and nothing at all where it was not or is not a mapping. """ flows: dict[str, Any] = self._mine().get("flows") or {} - kept: dict[str, Any] = flows.get(flow) or {} - held = kept.get(under) + kept = flows.get(flow) + if not isinstance(kept, dict): + return {} + held = cast("dict[str, Any]", kept).get(under) return cast("dict[str, Any]", held) if isinstance(held, dict) else {} def remember( self, flow: str, - names: tuple[str, ...], - models: Sequence[Runs], - config: dict[str, Any] | None = None, + agents: Mapping[str, Runs], + envs: Mapping[str, str] | None = None, + params: dict[str, Any] | None = None, budget: dict[str, Any] | None = None, ) -> None: """Writes down what this workspace is set up to run, so that it opens that way. Args: flow: The flow to run. - names: What that flow calls each agent it drives, which is "" apiece for a flow - that said how many it drives and nothing more. - models: What each of them runs and where, in the order the flow takes them. - config: What the flow itself was set up with, or None to leave whatever was kept - for it as it was -- choosing the agents again is not a way of forgetting how the - flow was set up. - budget: What a run of it here may spend, or None to leave whatever was kept as it - was. The same asymmetry as `config` and for the same reason: the flow's whole + agents: What each of its agent roles runs, by role. + envs: What each of its environment roles is given, as `-e` spells one, or None to + leave whatever was kept for it as it was -- choosing the agents again is not a + way of forgetting where they work. + params: What the flow itself was set up with, or None to leave whatever was kept. + budget: What a run of it here may spend, as JSON, or None to leave whatever was + kept. The same asymmetry as the rest and for the same reason: the flow's whole entry is replaced below, so what is not handed in has to be read back or it is - forgotten. A value that is empty erases it, which is how the menu says a flow is - back to running under whatever the flow itself says. + forgotten. A value that is empty erases it. """ - agents: dict[str, dict[str, Any]] = { - # By what the flow calls it, or by where it comes in the line when it has no name. - (names[at] if at < len(names) and names[at] else str(at + 1)): written(runs) - for at, runs in enumerate(models) - } mine = self._mine() mine["flow"] = flow - kept: dict[str, Any] = {"agents": agents} - held = config if config is not None else self.config(flow) - if held: - kept["config"] = held - spends = budget if budget is not None else self.budget(flow) - if spends: - kept["budget"] = spends + kept: dict[str, Any] = { + "agents": {role: written(runs) for role, runs in agents.items()} + } + for under, given in (("envs", envs), ("params", params), ("budget", budget)): + held = dict(given) if given is not None else self._kept(flow, under) + if held: + kept[under] = held mine.setdefault("flows", {})[flow] = kept self._write() diff --git a/src/hmz/sdk/__init__.py b/src/hmz/sdk/__init__.py index 49bc8c7d..ee131197 100644 --- a/src/hmz/sdk/__init__.py +++ b/src/hmz/sdk/__init__.py @@ -3,7 +3,7 @@ from hmz.sdk import Hmz hmz = Hmz() - hmz.run("chat", [], "say hello").run() + hmz.run("chat", "say hello", agents={"assistant": "claude/claude-haiku-4-5:low"}).run() There are two ways to reach a run from out here, and both are offered. @@ -18,6 +18,10 @@ started asks: what is being held here, what it is running, letting go of the terminals on it, stopping it. A run held that way outlives the program that asked for it. +:mod:`fakes` is the third thing a tool outside wants: the in-memory drivers a flow is tested +on -- scripted agents, dictionary filesystems, an outworlder that answers from a list -- which +are :mod:`hmz.runtime.flowing.fakes`, handed through whole. + Nothing under this names it. What is here is not a layer humanize is built out of -- every answer is written where it is carried out, in :mod:`hmz.runtime` and :mod:`hmz.daemon`, and this restates none of it. That is the whole of the difference between this and the seam every @@ -35,7 +39,17 @@ if TYPE_CHECKING: from hmz.daemon import Daemon, Held, Session - from hmz.runtime import Accounts, Epics, Fallbacks, Flows, Flowverses, Hmz, Run + from hmz.runtime import ( + Accounts, + Epics, + Fallbacks, + Flows, + Flowverses, + Hmz, + Refused, + Run, + ) + from hmz.runtime.flowing import fakes from hmz.sdk.daemons import Daemons __all__ = [ @@ -48,8 +62,10 @@ "Flowverses", "Held", "Hmz", + "Refused", "Run", "Session", + "fakes", ] #: Which front door each of them is behind: the runtime, reached straight, and the daemon @@ -67,10 +83,14 @@ "Flowverses": "hmz.runtime", "Held": "hmz.daemon", "Hmz": "hmz.runtime", + "Refused": "hmz.runtime", "Run": "hmz.runtime", "Session": "hmz.daemon", } +#: What is offered as a module rather than out of one: the fakes are a kit, used as one. +_MODULES = {"fakes": "hmz.runtime.flowing.fakes"} + def __getattr__(name: str) -> object: """Hands through what this package offers, out of the layer it is written in. @@ -87,6 +107,8 @@ def __getattr__(name: str) -> object: """ from importlib import import_module + if name in _MODULES: + return import_module(_MODULES[name]) where_ = _WRITTEN.get(name) if where_ is None: raise AttributeError(f"module {__name__!r} has no attribute {name!r}") diff --git a/src/hmz/tui/app.py b/src/hmz/tui/app.py index b72a905a..472efa86 100644 --- a/src/hmz/tui/app.py +++ b/src/hmz/tui/app.py @@ -43,7 +43,7 @@ from collections import deque from dataclasses import dataclass, field from pathlib import Path -from typing import TYPE_CHECKING, Any, ClassVar, NamedTuple, cast +from typing import TYPE_CHECKING, Any, ClassVar, NamedTuple, Protocol, cast import pyfiglet from rich.box import ROUNDED @@ -77,6 +77,7 @@ Adjusted, Adjusts, Chosen, + Declared, Drawn, Epics, Fallbacks, @@ -89,11 +90,10 @@ Reports, Runs, budget_of, - config_of, - dimensions, - model_of, - opens_on, - places_of, + declared_of, + named_as, + params_model, + params_of, reads, settled, ) @@ -102,13 +102,57 @@ from .tally import Tally if TYPE_CHECKING: - from collections.abc import Callable, Sequence + from collections.abc import Callable, Mapping, Sequence from pydantic import BaseModel from hmz.coganchor.agents import AgentBase, Board, Event, Question, SessionBase from hmz.daemon import Session - from hmz.runtime.flowing import Place + from hmz.flows import Budget, Usage + from hmz.runtime.flowing.harnesses import Listener + + +class _Running(Protocol): + """A run of a flow, as the interface holds one: `hmz.runtime.Run`, named by its shape. + + Named here rather than imported, the runtime being reached through the daemon alone. + """ + + @property + def budget(self) -> Budget: + """What the run may spend.""" + ... + + @property + def usage(self) -> Usage: + """What it has spent so far.""" + ... + + @property + def flow(self) -> str: + """The flow, as it was named.""" + ... + + def watch(self, listener: Listener) -> None: + """Has everything every session of the run says reach `listener`.""" + ... + + def opened(self, callback: Callable[[str, AgentBase, SessionBase], None]) -> None: + """Has each session the run opens told to `callback` as it opens.""" + ... + + def run(self) -> object: + """Runs the flow here, until it returns.""" + ... + + def stop(self) -> None: + """Stops the flow, which unwinds in its own time.""" + ... + + def close(self) -> None: + """Stops it and ends every conversation still open, without waiting.""" + ... + #: How often the right-hand column and the status line are redrawn, in seconds. _REFRESH = 0.5 @@ -147,13 +191,6 @@ #: prompt, and each of them one somebody has to work out for themselves. _UNWINDING = "it is closing out the turn it was in" -#: The three steps one agent of a flow is configured in, in the order they are asked: which -#: coding agent takes its turns and as whom, which model it runs and at what effort, and -- -#: only for a place the flow said may be pointed anywhere -- which machine its work lands on. -#: Each depends on the one before it: an account belongs to a backend, and a model belongs to -#: the CLI that runs it. -_WHO, _WHAT, _WHERE = 0, 1, 2 - #: The flow the interface opens on, which is the one that is only talking to one agent. _STARTS_ON = "chat" @@ -636,9 +673,10 @@ def action_quit(self) -> None: # pyright: ignore[reportIncompatibleMethodOverri A flow is a loop and a turn can think for minutes, so leaving without stopping it would leave the interface gone and the work going -- which reads as a hang. """ - for agent in self._agents: - agent.stop() - self._agents = [] + for run in (self._run, self._stopping): + if run is not None: + run.close() + self._run, self._stopping, self._agents = None, None, [] self._close_btw() self.exit() @@ -657,7 +695,7 @@ async def action_exit(self) -> None: who has decided to leave should be asked what becomes of it once rather than having to know which of two words asks. """ - if not self._agents: + if self._run is None: self.action_quit() return said = await self.push_screen_wait(Leaves(held=self._session is not None)) @@ -738,9 +776,9 @@ def action_interrupt(self) -> None: # would do has been gone for most of that. self._presses = self._presses + 1 if now - self._pressed < _AGAIN else 1 self._pressed = now - if self._agents: + if self._run is not None: self._interrupts() - elif self._stopping: + elif self._stopping is not None: # A flow told to stop and not yet gone. However long ago it was told: this is # the one thing left that a key can do about it, and asking for it twice over # would be asking twice about a run that is already over. @@ -764,52 +802,51 @@ def _interrupts(self) -> None: def _forces(self) -> None: """The press after the one that stopped a flow, which does not wait for it to go. - Telling an agent to stop closes its conversations, and a backend that took no notice - of that is the reason there is a third press: every conversation still open is closed - again, which is the backend's process going. What the flow gets back is a turn that - failed, the same thing it would have got had the agent fallen over by itself. + Stopping a flow interrupts the turn it was in and lets it unwind, and a backend that + took no notice of that is the reason there is a third press: every conversation still + open is closed, which is the backend's process going. What the flow gets back is a + turn that failed, the same thing it would have got had the agent fallen over by itself. And the run reads as over from here, whatever is still unwinding behind it: nothing is left that a turn could still be running in, so nothing is left spinning at the person who has just asked twice for it to stop. """ + run, self._stopping = self._stopping, None + self._presses = 0 + if run is None: + return closing = [ session - for agent in self._stopping + for agent in self._ran for session in agent.sessions if session in self._working ] - self._presses = 0 - self._stopping = [] - if not closing: - return - self.show( - f"[dim]— closing {len(closing)} conversation(s) under their turns —[/dim]" - ) + if closing: + self.show( + f"[dim]— closing {len(closing)} conversation(s) under their turns —[/dim]" + ) # On this thread, as telling the flow to stop is: closing a conversation is closing # the process behind it, which is a second at the outside. + run.close() for session in closing: - with contextlib.suppress(Exception): # a conversation already gone is gone - session.close() self._working.discard(session) def __init__( self, flow: str = "", - agents: Sequence[Runs] = (), - config: BaseModel | None = None, + agents: Mapping[str, Runs] | None = None, + params: BaseModel | None = None, session: Session | None = None, ) -> None: """Initializes an interface holding no agents, because nothing is running yet. Args: flow: The flow to open on, which is what a run being picked up names -- or "" to - open on what - this workspace was last set up to run, and on the one that only talks to one - agent where it has run nothing. - agents: What each of that flow's agents runs, in the order it takes them, or - nothing to open on what was remembered. - config: What that flow is set up with, or None to open on what was remembered. + open on what this workspace was last set up to run, and on the one that only + talks to one agent where it has run nothing. + agents: What each of that flow's agent roles runs, by role, or None to open on + what was remembered. + params: What that flow is set up with, or None to open on what was remembered. Checked by whatever read the line: an interface is opened set up, not corrected. session: What is holding this run somewhere a terminal closing cannot reach, or None for one opened in the terminal it is drawn on -- where letting go of the @@ -832,7 +869,13 @@ def __init__( #: that is only in the terminal it was opened in. The whole of what the interface #: knows about that: how many terminals are reading, and how to let go of them. self._session = session - #: The agents of the flow running now, which is who a typed line is said to. + #: The run going now, or None: which is what `a flow is running` means here. + self._run: _Running | None = None + #: Which run is the one going now, counted: what the person outside it is asked is + #: answered only while the run asking is still the one going. + self._generation = 0 + #: The agents behind the sessions the run going now has opened, which is who a typed + #: line is said to. One per session, each named for the role it was opened for. self._agents: list[AgentBase] = [] #: What the flow has done so far, which is what the right-hand column shows, and who #: reads the agents' own logs into it while it runs. @@ -857,29 +900,27 @@ def __init__( #: agents said to each other and to you is that, and a tool row per file read is a #: transcript nobody is reading and the answer scrolled off the top of it. self._details = False - #: Whether an agent may stop and ask, which `/afk` toggles. It may, until you say you - #: are not there: a question nobody answers is a flow that has stopped. + #: Whether anybody is here to be asked, which `/afk` toggles. They are, until you say + #: you are not: a flow that asks the person outside it and is answered by nobody is a + #: flow that has stopped. Away, the outworlder answers what an away one answers. self._afk = False - #: The question a turn has stopped on, if one has, and where its answer goes -- and + #: The question the flow has stopped on, if one has, and where its answer goes -- and #: which agent it was shown against, so that what it will take for an answer is shown #: under it rather than wherever the person is looking by the time it lands. self._asked_on: str | None = None self._asking: Question | None = None self._answer = "" self._answered = threading.Event() + #: The last thing a turn answered with, and the last thing an agent stopped to ask: + #: what the flow puts to the person is often exactly that -- a conversation says the + #: agent's answer back to you to ask what next -- and a line already on the screen is + #: not put there twice. + self._last_answer = "" + self._last_asked = "" #: When something was last copied off the screen, so that the status line can say so #: for a moment: a clipboard is written to silently, and a gesture that says nothing #: is one nobody knows worked. self._copied = 0.0 - #: The flow to run and what each of its agents runs, which start out as the flow that - #: is only talking to one agent and the first agent there is to talk to. So the first - #: thing you say starts something rather than being told to pick a flow first: a flow - #: is what you reach for once talking to one agent is not the shape of the work, and - #: nobody knows that before they have said anything. - #: - #: Nothing at all until some backend here has said what it runs, which is asked for in - #: the background as this opens: a model to open on is one of that CLI's own, and - #: there is no telling what those are without asking it. #: humanize, as the one object everything the interface does goes through: the #: flows there are, the agents and accounts they run as, the runs already made here #: and the run being started now. Reached through the daemon, which is the process a @@ -889,50 +930,51 @@ def __init__( #: What this workspace was last set up to run, so that opening it again finds it #: that way rather than back at the default. self.settings = self.hmz.settings + #: The flow to run and what each of its roles is given, which start out as the flow + #: that is only talking to one agent and the first agent there is to talk to. So the + #: first thing you say starts something rather than being told to pick a flow first: + #: a flow is what you reach for once talking to one agent is not the shape of the + #: work, and nobody knows that before they have said anything. self._flow_named = flow or self.settings.flow or _STARTS_ON - self._models = list(agents) - #: One place per agent the flow drives: what the flow calls it, which is "" apiece - #: for a flow that said how many it drives and nothing more, and the moments it needs - #: that one to run. Kept beside the models rather than read off the flow each time the - #: line above the prompt is drawn: that means loading and running a Python file, and - #: this is drawn twice a second. - self._wanted = self._places_of(self._flow_named) - # What is installed here and what each of them says it runs, which is what a place - # nothing was remembered for falls back on. - backends = installed() - if not self._models: - self._models = self.settings.agents( - self._flow_named, - tuple(place.goals for place in self._wanted), - ) or opens_on( - backends, - goals=self._wanted[0].goals if self._wanted else True, - ) - # If the flow would not load, `_places_of` falls back to agents already in hand; - # the remembered ones were not in hand on the first read. - if not self._wanted and self._models: - self._wanted = self._places_of(self._flow_named) - self._models = settled(self._models, self._wanted, backends) - #: What the flow itself is set up with, for a flow that says it can be set up at - #: all: an instance of the model it declared, or None. Read back from what this - #: workspace last ran, so a flow of many settings opens the way it was left. - self._config = config or config_of( - self._flow_named, self.settings.config(self._flow_named) + #: What the flow declares, kept rather than read off the flow each time the line + #: above the prompt is drawn: that means importing the flow, and this is drawn twice + #: a second. None for a flow that will not load. + self._declared: Declared | None = declared_of(self._flow_named) + # What is installed here and what each of them says it runs, which is what a role + # nothing was remembered for falls back on. Nothing at all until some backend here + # has said what it runs, which is asked for in the background as this opens. + remembered = ( + dict(agents) + if agents is not None + else self.settings.agents(self._flow_named) + ) + self._models: dict[str, Runs] = ( + settled(remembered, self._declared.agents, installed()) + if self._declared is not None + else remembered ) - #: What a run of it here may spend, or None for a flow nobody has set one for -- - #: which is a run under whatever the flow itself declares. Beside the config and not - #: inside it: it is a setting of the run rather than one of the flow's own. - self._budget = budget_of(self._flow_named) + #: Where each environment role is, as `-e` says it. + self._envs: dict[str, str] = self.settings.envs(self._flow_named) + #: What the flow itself is set up with, for a flow that takes params at all: an + #: instance of its model, or None. Read back from what this workspace last ran, so a + #: flow of many params opens the way it was left. + self._params: BaseModel | None = params or params_of( + self._flow_named, self.settings.params(self._flow_named) + ) + #: What a run of it here may spend, or None for none yet -- which only a flow + #: humanize ships runs with. Beside the params and not inside them: it is a setting + #: of the run rather than one of the flow's own. + self._budget: Budget | None = budget_of(self._flow_named) #: What has been typed here before, which the arrows walk. Read now rather than each #: time it is asked for: a run started here writes this project's own history into #: being, and what is being walked must not change under whoever is walking it. self.history = History() #: When each agent's turn started, for the line that closes it. self._began: dict[str, float] = {} - #: What each transcript has to show: one per agent, under that agent's id, and one - #: under `_EVERY` where all of them appear together. Kept by name rather than by the - #: object, since an agent's conversations come and go under it -- a Ralph loop opens - #: one a turn -- and what is read is the agent rather than whichever of them is open. + #: What each transcript has to show: one per role, under the role, and one under + #: `_EVERY` where all of them appear together. Kept by name rather than by the + #: object, since a role's sessions come and go under it -- a Ralph loop opens one a + #: turn -- and what is read is the role rather than whichever of them is open. self._kept: dict[str, _Kept] = {} #: Which transcript is being read: what the screen shows, and which agent a typed #: line is said to. The one they all appear on until somebody steps off it, that @@ -943,14 +985,13 @@ def __init__( #: counts for. self._pressed = 0.0 self._presses = 0 - #: The agents of a flow that has been told to stop and has not finished unwinding. - #: A third ctrl+c closes their conversations under whatever turn is still open, which - #: is the only thing left that a key can do about a flow already on its way out. - self._stopping: list[AgentBase] = [] + #: A run that has been told to stop and has not finished unwinding. A third ctrl+c + #: closes its conversations under whatever turn is still open, which is the only + #: thing left that a key can do about a flow already on its way out. + self._stopping: _Running | None = None #: The agents of the last run, which outlive it: their transcripts are still on the #: screen when the flow is over, so the diagram that reads one out is still about - #: them. `_agents` is the running flow's and is let go of the moment it ends, so that - #: the next thing typed starts something rather than being put to a flow that is gone. + #: them. The run going now's are the same list, filled as its sessions open. self._ran: list[AgentBase] = [] #: The conversations with a turn open, which are the only ones a typed line can go #: into: one written to a conversation between turns is answered on its own, outside @@ -974,33 +1015,44 @@ def __init__( self._spoke = threading.Event() self._awaiting = False - def _places_of(self, flow: str) -> tuple[Place, ...]: - """The agents a flow drives: what it calls each one, and what each has to be able to do. + def said(self) -> dict[str, Any]: + """What this interface says about the run it is holding, for the daemon's status. - Args: - flow: The flow, by name or as a path. + Called from the daemon's own thread, so it reads what the run keeps under its own + locks and nothing of the screen. As JSON: a budget with no limit on its cost is + written as the string `Infinity`, which reads back. Returns: - One place apiece -- and one unnamed place per agent already in hand for a flow that - will not load, since a name is a label on something that runs and not a reason for - anything to stop. + The flow, what the run may spend and what it has spent -- the two as None with + nothing running. """ - from hmz.runtime.flowing import Place - - # By the name it was chosen under, not by the file that name resolves to: a file may - # hold several flows, and which of them was asked for is the half after the colon -- - # which resolving the name to a path throws away. - places = places_of(flow) - if places is not None: - return places - return tuple( - Place(name="", person=False, moments=frozenset()) for _ in self._models - ) + import json + + run = self._run + return { + "flow": run.flow if run is not None else self._flow_named, + "budget": json.loads(run.budget.model_dump_json()) + if run is not None + else None, + "usage": json.loads(run.usage.model_dump_json()) + if run is not None + else None, + } @property def _named_by(self) -> tuple[str, ...]: - """What the flow calls each agent it drives, which is what a line about one says.""" - return tuple(place.name for place in self._wanted) + """The agent roles somebody chooses an agent for, in the order the flow declares them. + + What a line about one says, and what every transcript but the shared one is named + by. For a flow that will not load, the roles something was remembered for. + """ + if self._declared is not None: + return self._declared.roles + return tuple(self._models) + + def _in_order(self) -> list[Runs]: + """What each agent role runs, in the order the flow declares them.""" + return [self._models.get(role, Runs("")) for role in self._named_by] def compose(self) -> ComposeResult: """The transcript, the offers, the editor, the status. The width is the transcript's. @@ -1074,11 +1126,8 @@ async def _asks_what_runs(self) -> None: continue # Which may be the first model there is to open on, for an interface that opened # with nothing installed to talk to. - if not self._models: - self._models = opens_on( - installed(), - goals=self._wanted[0].goals if self._wanted else True, - ) + if self._declared is not None: + self._models = settled(self._models, self._declared.agents, installed()) self._draw() @work @@ -1127,7 +1176,7 @@ async def _freshens_flows(self) -> None: # package until this lands. if not one.url: continue - if self._agents or self._stopping: + if self._run is not None or self._stopping is not None: return if one.fetched and await asyncio.to_thread(verses.edited, one): continue @@ -1425,35 +1474,39 @@ def _driven(self) -> list[AgentBase]: is asked of the conversations, and there are none of those once a run is over. Returns: - One apiece, in the order the flow takes them, less the person -- who is talked to - at this prompt rather than read. + One per session opened, oldest first, each named for its role. """ - from hmz.coganchor.agents import HumanAgent + return list(self._agents or self._ran) - held = self._agents or self._ran - return [one for one in held if not isinstance(one, HumanAgent)] + def _of(self, role: str) -> list[AgentBase]: + """The agents behind every session one role has opened, oldest first.""" + return [one for one in self._driven() if one.id == role] def _working_agents(self) -> list[str]: - """Which of the flow's agents have a turn open, in the order the flow takes them. + """Which of the flow's roles have a turn open, in the order they opened sessions. Returns: - Their ids. These are the ones tab steps between: with ten agents going, what + Their names. These are the ones tab steps between: with ten agents going, what somebody is stepping between is the ones thinking. """ - return [ - agent.id - for agent in self._driven() - if any(session in self._working for session in agent.sessions) - ] + return list( + dict.fromkeys( + agent.id + for agent in self._driven() + if any(session in self._working for session in agent.sessions) + ) + ) def _reading(self) -> AgentBase | None: - """The agent being read, where one is rather than all of them. + """The agent being read, where one role is rather than all of them. Returns: - The agent, or None on the transcript they all appear on and for one whose flow is - over -- the transcript stays up either way, there being nothing to say to it. + The newest agent of that role, or None on the transcript they all appear on and + for one whose flow is over -- the transcript stays up either way, there being + nothing to say to it. """ - return next((one for one in self._driven() if one.id == self._attached), None) + held = self._of(self._attached) if self._attached != _EVERY else [] + return held[-1] if held else None def _says_to(self) -> SessionBase | None: """Which conversation a typed line goes into, which is the one on the screen. @@ -1496,10 +1549,9 @@ def _now_reading(self, whose: str, *, stepped: bool = True) -> None: # them: an agent left marked unread here would be marked for what is on the screen. for one in self._kept.values(): one.unread = False - agent = self._reading() shown = self.query_one("#transcript", Transcript) shown.clear() - held = len(agent.sessions) if agent is not None else 0 + held = sum(len(one.sessions) for one in self._of(whose)) if whose else 0 many = f"{_DOT}{held} conversations" if held > 1 else "" named = "every agent" if whose == _EVERY else f"{escape(short(whose))}{many}" shown.write( @@ -1522,22 +1574,31 @@ def _unread(self, whose: str) -> bool: return kept is not None and kept.unread def _held(self) -> list[Held]: - """How many conversations each of the flow's agents has, and which one is being read. + """How many conversations each of the flow's roles has, and which one is being read. Returns: - One per agent the flow drives, in the order it takes them -- and nothing at all - with no flow running, which is a line about what is set up rather than about what - it is doing. + One per agent role, in the order the flow declares them -- and nothing at all with + no flow run, which is a line about what is set up rather than about what it is + doing. """ - return [ - Held( - many=len(agent.sessions), - reading=agent.id == self._attached, - unread=self._unread(agent.id), - working=any(one in self._working for one in agent.sessions), + if not self._driven(): + return [] + held: list[Held] = [] + for role in self._named_by: + agents = self._of(role) + held.append( + Held( + many=sum(len(agent.sessions) for agent in agents), + reading=role == self._attached, + unread=self._unread(role), + working=any( + one in self._working + for agent in agents + for one in agent.sessions + ), + ) ) - for agent in self._driven() - ] + return held def action_attach_next(self) -> None: """Reads the next agent that is working, which is what tab is for.""" @@ -1661,12 +1722,12 @@ def _draw(self) -> None: # Left, first match wins, as opencode's status line resolves it: what is running if # anything is, else where this is. Right, the usage. The two ends are pushed apart. working = self._monitor.now_working() - if self._agents and not working and self._awaiting: + if self._run is not None and not working and self._awaiting: # A flow that has run out of things to do until it is told one. Spinning a bar at # it would read as a turn that has been thinking for as long as you have been # deciding what to say, which is the opposite of what is happening. left = f"[$text-muted]{_SPINNER[0]} waiting for you{_DOT}ctrl+c twice to stop[/]" - elif working or self._agents: + elif working or self._run is not None: bar = _SPINNER[int(time.monotonic() / _REFRESH) % len(_SPINNER)] # Whoever is talking and how long their turn has been going, or -- between two # turns -- the flow itself and how long the run has. A flow sleeps off a round, @@ -1712,8 +1773,8 @@ def _draw(self) -> None: # and they are read one at a time, against the name the flow calls each one by -- and # with the conversations each of them is holding, since one of those is what is being # read and what a typed line goes to. - lines = reads(self._named_by, self._models, self._held()) or [ - "no agent installed" + lines = reads(self._named_by, self._in_order(), self._held()) or [ + "no agent installed" if self._named_by else "no agent to choose" ] if spent: costing = f"{money(bill)}{floor}{_DOT}" if bill is not None else "" @@ -1771,17 +1832,18 @@ def _draw(self) -> None: def _flowing(self) -> str: """What is running now, flow inside flow, for the line that names one. - A flow may reach for another by name and run it, so what is running is a list rather + A flow may reach for another by ref and run it, so what is running is a list rather than a name: the one that was started, and whatever it called, innermost last. Read - from the runner rather than asked of the flow -- a flow is a Python file and may branch - any way it likes, so what it is doing is only ever visible where it was started. + from the running tree rather than asked of the flow -- a flow may branch any way it + likes, so what it is doing is only ever visible where it is being run. Returns: The flows, innermost last, and the one that is set up to run where none is running -- which is what this line says with nothing going on. """ return ( - " ▸ ".join(one.flow for one in self.hmz.flows.running()) or self._flow_named + " ▸ ".join(named_as(one.ref) for one in self.hmz.flows.running()) + or self._flow_named ) def _waiting_lines(self, beside: int = 0) -> list[str]: @@ -1884,7 +1946,7 @@ def _keys(self) -> list[str]: "enter answer" if self._asking is not None else "enter say" - if self._agents + if self._run is not None else "enter start" ) if len(self._ring()) > 1: @@ -1898,11 +1960,13 @@ def _keys(self) -> list[str]: keys.append("ctrl+c clear") elif self._counting(): keys.append( - "ctrl+c again to stop" if self._agents else "ctrl+c again to exit" + "ctrl+c again to stop" + if self._run is not None + else "ctrl+c again to exit" ) - elif self._agents: + elif self._run is not None: keys.append("ctrl+c stop") - elif self._stopping: + elif self._stopping is not None: keys.append("ctrl+c close them") else: keys.append("ctrl+c exit") @@ -1942,7 +2006,7 @@ def _mid_run(self, what: str) -> bool: Returns: True if a flow is running or on its way out, having said which. """ - if self._agents: + if self._run is not None: self.show( f"hmz: {what} while a flow is running: ctrl+c twice stops it first", "red", @@ -1952,7 +2016,7 @@ def _mid_run(self, what: str) -> bool: # its round, a turn is closed out -- and it writes down where it got to as it goes, # so a run picked up from a state that is still moving is a round done twice. And # `ctrl+c twice` is not the answer here: it has already been pressed. - if self._stopping: + if self._stopping is not None: self.show( f"hmz: {what} while the flow is still stopping: {_UNWINDING}", "red" ) @@ -1972,9 +2036,9 @@ async def action_monitor(self) -> None: Monitoring( self._flow_named, self._named_by, - self._models, + self._in_order(), self._monitor, - self._config, + self._params, drawn=self._boxes, reading=self._attached, board=self._board, @@ -2003,35 +2067,43 @@ def _board(self) -> Board | None: ) def _boxes(self) -> list[Drawn]: - """The agents that have worked, as the diagram draws them, in the flow's own order. + """The roles that have worked, as the diagram draws them, in the flow's own order. - The ones that have worked rather than the ones the flow declares. A flow may drive - ten agents and reach three of them, and seven boxes that have never done anything are - seven rows saying nothing -- the diagram is what the run *is doing*, and a flow is a - Python file that may never take the branch the other seven are on. Each appears as + The ones that have worked rather than the ones the flow declares. A flow may declare + ten roles and reach three of them, and seven boxes that have never done anything are + seven rows saying nothing -- the diagram is what the run *is doing*. Each appears as its first turn starts and stays for the rest of the run, which is what makes this a picture of the run growing rather than a list of what was configured. Returns: - One per agent that has taken a turn, in the order the flow takes them, and nothing - at all before the first turn of a run -- which is a sheet about what is set up - rather than about what it is doing. + One per role that has taken a turn, in the order the flow declares them, and + nothing at all before the first turn of a run -- which is a sheet about what is set + up rather than about what it is doing. """ - named = self._named_by shape = self._monitor.shape() - return [ - Drawn( - who=agent.id, - named=named[at] if at < len(named) else "", - runs=self._models[at].spec if at < len(self._models) else "", - working=working, - reading=agent.id == self._attached, - unread=self._unread(agent.id), + named = self._named_by + seen = list(dict.fromkeys(agent.id for agent in self._driven())) + seen.sort(key=lambda who: named.index(who) if who in named else len(named)) + drawn: list[Drawn] = [] + for who in seen: + working = any( + one in self._working + for agent in self._of(who) + for one in agent.sessions ) - for at, agent in enumerate(self._driven()) - if (working := any(one in self._working for one in agent.sessions)) - or shape.turns.get(agent.id, 0) - ] + if not (working or shape.turns.get(who, 0)): + continue + drawn.append( + Drawn( + who=who, + named=who, + runs=self._models[who].spec if who in self._models else "", + working=working, + reading=who == self._attached, + unread=self._unread(who), + ) + ) + return drawn @on(Editor.Sent) def _sent(self, event: Editor.Sent) -> None: @@ -2126,7 +2198,7 @@ def action_btw(self, question: str = "") -> None: if not question: self.show("hmz: usage: /btw ", "red") return - if not self._agents: + if self._run is None: self.show("hmz: /btw needs a flow that is running", "red") return candidates = self._btw_candidates() @@ -2171,24 +2243,18 @@ def action_btw(self, question: str = "") -> None: def _btw_snapshot(self) -> FlowSnapshot: """Copies the current run into a prompt-sized, immutable observation.""" - from hmz.coganchor.agents import HumanAgent - shape = self._monitor.shape() - driven = tuple(self._agents) - named = self._named_by + # One per role, however many sessions it opened: a role is what is watched, and each + # of its sessions is an agent of its own named for it. + driven = {agent.id: agent for agent in self._agents if agent.id} agents = tuple( - ( - at, - AgentProgress( - agent=agent.id, - model=agent.config.model, - turns=shape.turns.get(agent.id, 0), - working=agent.id in shape.working, - ), + AgentProgress( + agent=who, + model=agent.config.model, + turns=shape.turns.get(who, 0), + working=who in shape.working, ) - for at, agent in enumerate(driven) - if not isinstance(agent, HumanAgent) - if agent.id + for who, agent in driven.items() ) handovers = tuple( sorted( @@ -2219,16 +2285,16 @@ def _btw_snapshot(self) -> FlowSnapshot: (one.kind, one.tokens, one.whole) for one in self._monitor.reckoning(now=ended or moment) ) - # Keep the role separate from the stable id used by the monitor and handover records. + # The role beside the id the monitor and the handovers use, which is the same word. labelled = tuple( AgentProgress( agent=item.agent, model=item.model, turns=item.turns, working=item.working, - role=named[index] if index < len(named) else "", + role=item.agent, ) - for index, item in agents + for item in agents ) return FlowSnapshot( flow=self._flowing(), @@ -2438,9 +2504,9 @@ def action_stop(self) -> None: then unwound, that is the interface closing on one key after a line that said there was nothing to stop. """ - if self._agents: + if self._run is not None: self.action_stop_flow() - elif self._stopping: + elif self._stopping is not None: self.show(f"hmz: the flow is already stopping: {_UNWINDING}", "red") else: self.show("hmz: no flow is running, so there is nothing to stop", "red") @@ -2450,26 +2516,27 @@ def action_stop(self) -> None: def action_stop_flow(self) -> None: """Stops the whole flow, not just the turn -- which is the second ctrl+c or `/stop`. - Every agent is told to take no further turn, so the one running now is closed out and - the loop driving it ends rather than handing on to the next agent. The agents are let - go of here rather than when the flow's own thread notices, so that the next thing - said starts something instead of being put to a flow that is on its way out -- and - kept as the ones stopping, since a flow unwinds in its own time and the press after - this one is the one that does not wait for it. + The turn running now is interrupted and every call of the flow unwinds from where it + stands, closing what it opened. The run is let go of here rather than when its own + thread notices, so that the next thing said starts something instead of being put to + a flow that is on its way out -- and kept as the one stopping, since a flow unwinds in + its own time and the press after this one is the one that does not wait for it. Silent when nothing is running, every caller having its own answer for that: the key is mid-gesture and the press after it says what it does, a flow chosen while none runs has nothing to say about the one that was not there, and `/stop` looks before it calls this and says for itself that there was nothing to stop. """ - for agent in self._agents: - agent.stop() - if self._agents: - self.show("[dim]— stopping the flow —[/dim]") + run = self._run + if run is None: + return + run.stop() + self.show("[dim]— stopping the flow —[/dim]") # Held by identity, so that the run's own thread can say when it has finished # unwinding and nothing says it of a run that started since. - self._agents, self._stopping = [], self._agents + self._run, self._stopping, self._agents = None, run, [] self._spoke.set() # and a flow waiting to be told hears that it is over + self._answered.set() # as does one waiting on an answer self._never_sent("the flow stopped first") def on_unmount(self) -> None: @@ -2480,10 +2547,12 @@ def on_unmount(self) -> None: waiting on a prompt that is not there, holding a backend open behind it. Said to nobody rather than to the transcript, which has gone with everything else. """ - for agent in self._agents: - agent.stop() - self._agents, self._stopping = [], [] + for run in (self._run, self._stopping): + if run is not None: + run.close() + self._run, self._stopping, self._agents = None, None, [] self._spoke.set() + self._answered.set() self._close_btw() def _never_sent(self, because: str) -> None: @@ -2536,7 +2605,7 @@ async def action_flow(self, named: str = "") -> None: Args: named: A flow of your own, as a path, to open the menu already holding. """ - running = bool(self._agents) + running = self._run is not None if named and running: self.show("hmz: a flow is running; no choosing a flow", "red") return @@ -2557,8 +2626,8 @@ async def _chooses(self, named: str, *, running: bool) -> Chosen | None: running: Whether a flow is running, which is what takes the flows away. Returns: - The flow, its agents and how the flow itself is set up, or None for a menu walked - out of -- which changes nothing at all. + The flow, what its roles are given and how the flow itself is set up, or None for a + menu walked out of -- which changes nothing at all. """ # Opened whether or not there is a backend to run one on: which flow to run is worth # reading either way, and the sheet an agent is set up on says for itself that there @@ -2569,14 +2638,15 @@ async def _chooses(self, named: str, *, running: bool) -> Chosen | None: # What is in hand is what is in hand for the flow the interface is set up on. A menu # opened straight into another flow is handed none, and reads what that one was last # set up with here -- which is what turning to it would have read. - holding = self._models if not named or named == self._flow_named else () + holding = not named or named == self._flow_named return await self.push_screen_wait( Flows( named or self._flow_named, - holding, - self._config if holding else None, + self._models if holding else {}, + self._params if holding else None, agents, self.settings.flows(), + envs=self._envs if holding else None, budget=self._budget if holding else None, unavailable=frozenset(unavailable), running=running, @@ -2612,7 +2682,7 @@ async def _quick_flow(self, named: str, task: str) -> None: telemetry.snag("unknown-flow", length=len(named)) self.show(f"hmz: no such flow: {named}", "red") return - if self._agents: + if self._run is not None: # The same answer `/flow ` gives while one runs, since it is the same thing # being asked for: two ways of choosing a flow that did opposite things would be # one of them ending a day's work on a line meant to queue the next one up. @@ -2635,98 +2705,82 @@ def _remembered_for(self, flow: str) -> Chosen | None: flow: The flow, by the name it is offered under. Returns: - The flow, its agents and how it is set up -- exactly what the menu would have been - saved holding -- or None for a flow to put that menu up about: one this workspace - has never set up, one that has grown, lost or renamed an agent since it last was, - and one whose kept settings no longer read back through the model it declares now. - A settings file is a convenience, and one that no longer fits the flow is a question - to ask again rather than a run to start on half an answer. A flow nothing was kept - for is not one of those: it takes its own defaults, exactly as it does on a command - line with nothing handed to it -- and neither is one that has since dropped its - settings altogether, which is a flow nothing is asked about. + The flow, what its roles are given and how it is set up -- exactly what the menu + would have been saved holding -- or None for a flow to put that menu up about: one + this workspace has never set up, one that has grown, lost or renamed a role since + it last was, one whose kept params no longer read back through the model it + declares now, and one given no budget that needs one. A settings file is a + convenience, and one that no longer fits the flow is a question to ask again rather + than a run to start on half an answer. """ - places = places_of(flow) - if places is None: - # A flow that will not load says nothing about what it drives, so nothing here can - # tell whether it is set up. Running it is where that is said, exactly as it is - # for the flow already in force. - return Chosen( - flow, tuple(self.settings.agents(flow)), budget=budget_of(flow) - ) - held = self.settings.flows().get(flow) - kept = cast("dict[str, Any]", held) if isinstance(held, dict) else {} - agents = kept.get("agents") - # By what the flow calls each place and in the flow's own order, which is how they - # were written down: a flow that grew a reviewer in the middle would otherwise read as - # set up and hand the builder's model to it. - wanted = [place.name or str(at + 1) for at, place in enumerate(places)] - if ( - not isinstance(agents, dict) - or list(cast("dict[str, Any]", agents)) != wanted - ): + declared = declared_of(flow) + agents = self.settings.agents(flow) + envs = self.settings.envs(flow) + if declared is None: + # A flow that will not load says nothing about what it declares, so nothing here + # can tell whether it is set up. Running it is where that is said, exactly as it + # is for the flow already in force. + return Chosen(flow, agents, envs, budget=budget_of(flow)) + if set(agents) != set(declared.roles): + return None + if any(role.required and role.name not in envs for role in declared.envs): return None - runs = self.settings.agents(flow, [place.goals for place in places]) - if len(runs) != len(places): - return None # written by hand, or written by something that writes it otherwise - written_ = self.settings.config(flow) - config = config_of(flow, written_) - if written_ and config is None and model_of(flow) is not None: - # Set up with settings this flow no longer accepts, which is one that has - # dropped, renamed or retyped a setting since. Nothing here can guess what the - # answer that no longer reads was meant to say, so it is asked where it is asked. - # A flow that dropped its settings model outright asks nothing, so it is not one - # of these: what was kept for it is a dead entry rather than a wrong answer. + written_ = self.settings.params(flow) + params = params_of(flow, written_) + if written_ and params is None and params_model(flow) is not None: + # Set up with params this flow no longer accepts, which is one that has dropped, + # renamed or retyped a param since. Nothing here can guess what the answer that no + # longer reads was meant to say, so it is asked where it is asked. return None - # Through the same settling every other way into the models goes through, or a flow - # that has since declared it needs the backend's own goals at a place would run here - # with them off and be refused before its first turn. - # What a run of it may spend, among the rest of what was remembered: this is the - # path a flow runs by without the menu ever opening, and one that dropped the - # allowance would start an unbounded run out of a workspace whose settings say six - # hours -- with nothing asked, the question living on the menu that did not open. - return Chosen(flow, tuple(settled(runs, places)), config, budget_of(flow)) + budget = budget_of(flow) + if budget is None and not declared.unbounded: + return None + return Chosen(flow, agents, envs, params, budget) def _took_flow(self, chosen: Chosen, *, running: bool, starting: str = "") -> None: """Applies what the flow menu was saved with, and writes it down. Args: - chosen: The flow, its agents, and how the flow itself is set up. + chosen: The flow, what its roles are given, and how the flow itself is set up. running: Whether a flow was running when the menu opened, which is what decides between starting fresh and changing the agents under a run. starting: What to start the flow on now that it is set up, for a `$` line that named the flow and said what to do in one go, or "" to leave it waiting to be told -- which is what every other way of choosing a flow leaves it doing. """ - places = places_of(chosen.flow) - same = (chosen.flow, list(chosen.agents), chosen.config, chosen.budget) == ( - self._flow_named, - self._models, - self._config, - self._budget, - ) + import json + + same = ( + chosen.flow, + chosen.agents, + chosen.envs, + chosen.params, + chosen.budget, + ) == (self._flow_named, self._models, self._envs, self._params, self._budget) if not running and not same: # A flow is chosen in order to be run, so whatever is running stops: the interface # opens on one already, and a choice that quietly went to the back of the queue # behind it would read as no choice at all. Answering the same way twice is not a # choice, though, and must not end the conversation. self.action_stop_flow() - self._flow_named, self._models = chosen.flow, list(chosen.agents) - self._wanted = places if places is not None else self._places_of(chosen.flow) - self._config = chosen.config - self._budget = chosen.budget + # Read again whether or not it is the same flow: a fetch or an edit since may have + # given it roles the menu was just saved with. + self._declared = declared_of(chosen.flow) + self._flow_named = chosen.flow + self._models, self._envs = dict(chosen.agents), dict(chosen.envs) + self._params, self._budget = chosen.params, chosen.budget self.settings.remember( chosen.flow, - self._named_by, self._models, - chosen.config.model_dump(mode="json") - if chosen.config is not None + self._envs, + # As JSON, which is what a settings file holds and reads back through the model. + json.loads(chosen.params.model_dump_json()) + if chosen.params is not None else None, - # An allowance that caps nothing is written down as nothing rather than as three - # zeros, which is how the menu says a flow is back to running under whatever the - # flow itself declares. Three zeros kept would override the flow's own default for - # good, and there would be no way left to say `as the flow has it`. - dimensions(chosen.budget) - if chosen.budget is not None and chosen.budget.bounded + # And a budget of nothing written down as nothing, which is how a flow that + # needs none is told apart from one that was given one. + json.loads(chosen.budget.model_dump_json()) + if chosen.budget is not None else {}, ) if running: @@ -2739,69 +2793,17 @@ def _took_flow(self, chosen: Chosen, *, running: bool, starting: str = "") -> No self._starts(starting) def _reconfigured(self) -> None: - """Sets the agents of a run that is going to what they have just been changed to. - - An agent is configured once and read from there on, so what a running one is doing now - is what it was set up with, and what it is asked for next is what it is set up with by - the time it is asked. So the ones whose CLI has not changed are set up where they - stand: the turn under way finishes as it started -- a model does not think harder - halfway through an answer -- and everything asked for after it is at the new model, - effort, account, rung and machine. - - A CLI that has changed is not one of those. What drives a backend is the class the - agent is, and the flow is holding the agents it was handed when it started; one of - them cannot become another backend without becoming another object, which is a thing - only starting the flow again does. So that one is written down and runs from the next - time the flow is started. - """ - from dataclasses import replace - - from hmz.coganchor.agents import anchored + """Says what becomes of agents changed under a run that is going. - for at, agent in enumerate(self._agents): - if at >= len(self._models): - break - runs = self._models[at] - cli, _, rest = runs.spec.partition("/") - model, _, effort = rest.rpartition(":") - if cli != agent.backend: - self.show( - f"[dim]{escape(agent.id)} is {escape(cli)} from the next run; " - "an agent cannot become another backend under the flow holding it[/dim]" - ) - continue - try: - machine = anchored(runs.anchor) - except ValueError as why: # a target that cannot be read is one to correct - self.show(f"hmz: {escape(agent.id)}: {why}", "red") - continue - # Said to the agent rather than to a session: what a person changes here they - # change about the agent, and every conversation it opens from now on is at it. - agent.reconfigure( - replace( - agent.config, - model=model, - effort=effort, - machine=machine, - provider=runs.provider, - goals=runs.goals, - # Both of the settings a sheet may leave unanswered are passed on only - # where it answered them: what is not said here is what the agent was - # configured with, and writing a `True` over it would be the interface - # switching searching on for an agent nobody asked about -- on a CLI - # that had it off, silently, every time somebody reopened this sheet. - **( - {"web_search": runs.web_search} - if runs.web_search is not None - else {} - ), - **({"permission": runs.permission} if runs.permission else {}), - ) - ) - self.show( - f"[dim]{escape(agent.id)} is {escape(runs.spec)} " - "from its next turn[/dim]" - ) + A run is handed a driver per role as it starts -- the CLI, the account, the model and + the effort -- and every session of that role is opened on it for as long as the run + goes. Nothing under a running flow can be swapped for another without the flow + noticing, so what was changed is written down and is what the next run starts on. + """ + self.show( + "[dim]the roles of the run going now stay as it started; what was changed is " + "what the next run starts on[/dim]" + ) @work async def _asks_about_reports(self) -> None: @@ -2924,7 +2926,7 @@ async def action_epics(self) -> None: is going, on the sheet where it was asked for. Whether one is going is asked there rather than handed over, since this list outlives the run it was opened during. """ - said = await self.push_screen_wait(Epics(running=lambda: bool(self._agents))) + said = await self.push_screen_wait(Epics(running=lambda: self._run is not None)) if said is None: return for one in said.said: @@ -2933,16 +2935,17 @@ async def action_epics(self) -> None: self._carries_on(said.epic) def action_resume(self, argv: Sequence[str] = ()) -> None: - """Carries the last run in this directory on, which is what `/resume` is for. + """Carries the last run here of a flow that can be picked up on: what `/resume` is for. `/epics` already offers this of whichever run you go into, and needing to find that row is the whole of what is wrong with it: a loop is left running overnight, the machine goes down, and what somebody who comes back to a stopped one wants is the - work carried on rather than a list to look for it in. So this is the last run here - and no other -- there is nothing to choose, which is why it is a command rather than - a row -- and a run that cannot be carried on says why rather than quietly handing the - one before it over: a loop resumed from the day before yesterday because yesterday's - died early is a day's work thrown away without anybody being told. + work carried on rather than a list to look for it in. So this is the last run here of + a resumable flow and no other -- a conversation had since is not a run to carry on, + and there is nothing to choose, which is why it is a command rather than a row -- and + a run that cannot be carried on says why rather than quietly handing the one before + it over: a loop resumed from the day before yesterday because yesterday's died early + is a day's work thrown away without anybody being told. Args: argv: Whatever was typed after the command, which is nothing: said back rather @@ -2956,17 +2959,32 @@ def action_resume(self, argv: Sequence[str] = ()) -> None: "red", ) return - runs = self.hmz.epics.all() # oldest first, so the last of them is the last run + epics = self.hmz.epics + runs = epics.all() # oldest first, so the last of them is the last run if not runs: self.show( "hmz: no flow has been run here, so there is nothing to carry on from", "red", ) return + # Past every run of a flow that neither was nor is one to pick up -- a conversation + # had since -- and no further: a run of one that was or is, or a record that cannot + # be read, is the one this settles on, and says for itself what stands in its way. + for epic in reversed(runs): + ran = epics.read(epic) + if ran is None or ran.resumable or self._picks_up(ran.flow): + break + else: + self.show( + "hmz: no run here was of a flow that can be picked up, so there is nothing " + "to carry on from", + "red", + ) + return # Which run is the whole of what this command settles. Why a run cannot be carried # on is settled in one place for both ways in, so that a run walked into on `/epics` # is turned down for the same reasons in the same words. - self._carries_on(runs[-1]) + self._carries_on(epic) def _picks_up(self, flow: str) -> bool: """Whether one flow says now that it can be picked up. @@ -2990,22 +3008,20 @@ def _picks_up(self, flow: str) -> bool: return False def _carries_on(self, epic: Path) -> None: - """Runs the flow of one run again, on what that run left behind. + """Runs the flow of one run again, picking up what that run left behind. Which is a run of its own: an epic is one run and is never reopened, so this is the - flow started again with the state of the run being picked up, writing into an epic of - its own that says which one it came from. + flow started again from the journal of the run being picked up, writing into an epic + of its own that says which one it came from. - The flow, its agents and what it was asked to do all come from the run rather than - from what the interface happens to be set up on: picking up a run means running what - ran, and an agent swapped under it would be a different run wearing its name. + The flow, what its roles were given, its params, its budget and what it was asked to + do all come from the run rather than from what the interface happens to be set up + on: picking up a run means running what ran, and an agent swapped under it would be + a different run wearing its name -- and one the journal would not pick up from. What stands in the way of carrying one on is read here and nowhere else, whether the run was named by `/resume` or walked into on `/epics`: two ways in that turned the - same run down for different reasons -- or one that took up what the other refused -- - would be two answers to one question. The flow is therefore asked again here, having - been asked to draw the row: a flow is a file, and the one that matters is the one it - is when somebody presses the key rather than when the list was drawn. + same run down for different reasons would be two answers to one question. Args: epic: The run to pick up, by the directory it is written in. @@ -3029,46 +3045,46 @@ def _carries_on(self, epic: Path) -> None: "red", ) return - # Nothing left behind is a run that stopped before it wrote down where it had got - # to, or one that emptied what it had written -- which is a flow saying the next run - # here starts clean. Either way carrying it on would be a run starting from the top - # wearing a line that says which run it came from, which is a record of something - # that did not happen. So it says what the next move is instead. - if not self.hmz.epics.state(epic, ran.flow): + # A journal with nothing in it is a run killed before it wrote down where it had got + # to, and carrying it on would be a run starting from the top wearing a line that + # says which run it came from -- a record of something that did not happen. + if not self.hmz.epics.picks_up(epic): self.show( f"hmz: {escape(ran.name)} left nothing behind, so there is nothing to " "carry on from: say what to do and the flow starts from the top", "red", ) return - # The person at the prompt is not one of the agents anybody chooses, so a flow that - # talks to one wrote down an agent nothing on a command line names -- and the run - # itself is what says which of them that was. # Named here rather than at the top of the file: `backends` is a local elsewhere in # this class, and an agent at no rung has to be written back out as `auto` or the # spec it goes into is `MODEL:`, which nothing can read again. from hmz.coganchor import backends + from hmz.flows import Budget - drove = [one for one in ran.agents if not one.person] self._flow_named = ran.flow - self._models = [ - Runs( + self._declared = declared_of(ran.flow) + self._models = { + one.agent: Runs( f"{one.backend}/{one.model}:{backends.written(one.effort)}", - permission=one.permission, - provider=one.provider, - goals=one.goals, + one.provider, ) - for one in drove - ] - self._wanted = self._places_of(ran.flow) - self._config = config_of(ran.flow, self.settings.config(ran.flow)) - self._budget = budget_of(ran.flow) - named = [part for runs in self._models for part in ("-a", runs.spec)] + for one in ran.agents + } + self._envs = { + role: spec + for role, _, spec in (one.partition("=") for one in ran.envs) + if spec + } + self._params = params_of(ran.flow, ran.params) + try: + self._budget = Budget.model_validate(ran.budget) if ran.budget else None + except ValueError: + self._budget = None self.show( f"[dim]carrying on from {escape(ran.name)}: {escape(ran.flow)} on what that " "run left behind[/dim]" ) - self._flow(["-f", ran.flow, *named, ran.task], resume=epic) + self._flow(ran.task, resume=epic) @work async def action_fallback(self) -> None: @@ -3156,7 +3172,9 @@ def _on_screen( try: asyncio.get_running_loop() except RuntimeError: # no loop here, so this is a thread of somebody's own - with contextlib.suppress(RuntimeError): # or the interface has gone + if not self.is_running: + return # and one that has gone has nothing left to draw on + with contextlib.suppress(RuntimeError): # or it went just now self.call_from_thread(lambda: doing(*said, **and_so)) return if self.is_running: # and one that has gone has nothing left to draw on @@ -3172,162 +3190,51 @@ def _went(self, held: list[str]) -> None: self._said_by_you(said) self._draw() - def _listen(self, agent: AgentBase) -> str | None: - """Waits at the prompt for a flow that has nothing to do until it is told something. - - Called from the flow's own thread, which waits here. Nothing on the event loop is - touched, so the interface goes on being an interface while a flow waits in it. - - Asked of the agent that is waiting rather than of whatever is running now: a flow - that has been stopped takes a while to unwind, and one still sitting here when the - next flow has started would otherwise read that flow's agents as its own -- and take - the line meant for it. + def _flow(self, task: str, resume: Path | None = None) -> None: + """Starts the flow that is set up, keeping the run so that a typed line reaches it. Args: - agent: Whose flow is waiting, which is the one this answers about. - - Returns: - What was said next, or None once this flow is over -- stopped by hand, or the - interface going away, either of which has to release this rather than leave a - thread waiting on a prompt that is not there. - """ - if agent.stopped or agent not in self._agents: - return None - self._awaiting = True - try: - while True: - # Cleared before the queue is read, so that a line arriving between the two - # sets it again and is not waited through. - self._spoke.clear() - if agent.stopped or agent not in self._agents: - return None - if held := self._take(): - # Whatever turn this answer starts is that line's turn, and takes - # nothing else out of the queue on the way in. - with self._saying: - self._handed = True - return "\n\n".join(held) - self._spoke.wait(_REFRESH) - finally: - self._awaiting = False - - def _as_they_were_set_up(self, chosen: list[AgentBase]) -> list[AgentBase]: - """Sets each agent up as it was chosen: where it works, what it holds, who it is. - - Done to the agents rather than said on the line that made them: all of them are - settings of the agent, and `hmz exec` reads a line that says what each one runs and - nothing else. An agent that works here and runs as this machine is signed in is left - exactly as it was. - - Args: - chosen: The agents the line named, in the order the flow takes them. - - Returns: - The same agents, or one set up in place of any that was given a machine, a rung of - what it may do, or an account to run as. - - Raises: - ValueError: If a target cannot be read, or an agent was given an account there is - no such thing as -- both before any of them has run, since either is a line to - correct at the prompt rather than a traceback out of a flow's own thread. + task: What it is to do. + resume: The run to pick up, for a flow that says it can be picked up, or None for + a run from the top. """ - from dataclasses import replace - - from hmz.coganchor.agents import anchored - - moved: list[AgentBase] = [] - for at, agent in enumerate(chosen): - runs = self._models[at] if at < len(self._models) else Runs("") - if ( - not runs.anchor - and not runs.permission - and not runs.provider - and agent.config.goals is runs.goals - and ( - runs.web_search is None - or agent.config.web_search is runs.web_search - ) - ): - moved.append(agent) - continue - if ( - runs.provider - and self.hmz.accounts.find(agent.backend, runs.provider) is None - ): - # Asked now rather than when the first turn needs it: an agent that cannot - # find the account it was told to run as must not quietly run as whoever - # started it is signed in as, and must not do it half an hour in. - raise ValueError( - f"no {agent.backend} provider called {runs.provider!r}" - ) - # The config is frozen, so an agent that works elsewhere, allowed less than an - # agent nobody asked about, or signed in as somebody else, is another agent at - # the same model and effort -- which is what it is. - moved.append( - type(agent)( - replace( - agent.config, - machine=anchored(runs.anchor), - provider=runs.provider, - goals=runs.goals, - # Said only where the sheet said it, for the reason the rung beside - # it is: an agent nobody was asked about is one this says nothing - # about, and the config keeps what it was made with. - **( - {"web_search": runs.web_search} - if runs.web_search is not None - else {} - ), - **({"permission": runs.permission} if runs.permission else {}), - ) - ) - ) - return moved - - def _flow(self, argv: list[str], resume: Path | None = None) -> None: - """Starts a flow, keeping its agents so that a typed line can reach one. - - Args: - argv: The command line, as `hmz exec` takes it. - resume: The run to pick up from, for a flow that says it can be picked up, or None - for one starting from whatever the last run of it here left -- which is what - running a resumable flow again means. - """ - if self._agents: + if self._run is not None: self.show("hmz: a flow is already running", "red") return + from hmz.runtime.flowing import open_outworlder + from hmz.runtime.kept import written + + self._generation += 1 + generation = self._generation try: - # `--json` says how a run is written for whoever is at a command line, and there - # is nobody at one here: the interface draws the same events itself. - path, chosen, task, _, _, _ = self.hmz.read(argv) - except SystemExit: - return # argparse has already said what was wrong, and it went to the transcript - try: - chosen = self._as_they_were_set_up(chosen) - except ValueError as why: # a target that cannot be read is a line to correct - self.show(f"hmz: {why}", "red") - return - try: - # Loaded here rather than on the thread it will run on, so that the agents it - # drives are in hand before anything is hooked up to them: a flow that says it - # talks to the person drives one more than was chosen, and the person is reached - # through this interface like everything else. How the flow itself is set up - # goes with them: it is a setting of the flow rather than of any agent, so it - # is not on the line that says what each of them runs. - runner = self.hmz.runner( - path, chosen, self._config, resume=resume, budget=self._budget + # Whoever is outside the run is whoever is at this prompt: asked on a thread of + # the run's own, and away while `/afk` says so -- which a run asks as it asks. + run: _Running = self.hmz.run( + self._flow_named, + task, + agents={ + role: written(runs) + for role, runs in self._models.items() + if role in self._named_by + }, + envs={ + role: spec + for role, spec in self._envs.items() + if self._declared is None or role in self._declared.places + }, + params=self._params.model_dump() if self._params is not None else None, + budget=self._budget, + resume=resume if resume is not None else False, + outworlder=open_outworlder( + ask=functools.partial(self._outworlder_asks, generation), + away=lambda: self._afk, + ), ) - except Exception as why: # noqa: BLE001 -- a flow that will not load is a line to fix + except Exception as why: # noqa: BLE001 -- a flow that will not start is a line to fix self.show(f"hmz: {why}", "red") return - # What a place declared that its agent's backend had no way of carrying, for a place - # that said it would rather run than be refused. Drawn where the interface's own - # lines go, before the run starts: the setting was dropped, and a drop nobody was - # told about would be the very thing the drop was there to avoid. - for line in runner.unserved().splitlines(): - self.show(f"hmz: {line}", "yellow") - agents = list(runner.agents) - self._agents = self._ran = agents + agents: list[AgentBase] = [] + self._run, self._agents, self._ran = run, agents, agents with self._btw_lock: old_side_sessions = [session for _, session in self._btw_active.values()] self._btw_active.clear() @@ -3341,65 +3248,56 @@ def _flow(self, argv: list[str], resume: Path | None = None) -> None: # Nothing is left of the flow before this one to press a key about, and what is # being read is one of its agents unless it was the transcript they all appear on. # Which is where a run is watched from, so it is where a run starts. - self._stopping = [] + self._stopping = None if self._attached != _EVERY: self._now_reading(_EVERY, stepped=False) self._monitor = Monitor() # What the run costs is read from the logs the agents keep, which they write as they # go: a backend only says what a turn cost once the turn is over, and a turn is long. - self._tally = Tally(agents, self._monitor) + self._tally = Tally([], self._monitor) self._tally.watch() with self._saying: self._queued, self._given, self._handed = [], [], False - - from hmz.coganchor.agents import HumanAgent - - for agent in agents: - agent.watch(self._heard) - if not isinstance(agent, HumanAgent): - # What its backend counts, said before its first turn: a kind nothing was - # spent on this turn is missing from that turn's reckoning exactly as a kind - # the CLI never counts is, and what is drawn of a run driving two backends has - # to tell the two apart to say which of its figures are whole. The person is - # not one of these -- nobody counts what a person costs -- and counting them - # as a backend that reports nothing would mark every figure of a run they are - # in as short of tokens nobody ever spent. What the CLI's own log says is - # added to this by the tally, once it has actually read one: Codex's server - # never names a cached read and the rollout it writes does, but a rollout - # written on another machine is one nothing here reads. - self._monitor.reporting(agent.id, type(agent).counts) - # Whichever turn starts next takes the oldest line that was held. - agent.waiting = self._at_turn_start - # Bound to the agent, so that each of these answers about the flow that is - # asking rather than about whichever flow is running by the time it is asked. - agent.ask = functools.partial(self._ask, agent) - agent.prompting = functools.partial(self._listen, agent) - self._draw() - - # This run's, whatever is being watched by the time it ends. watching, tally = self._monitor, self._tally + run.watch(self._heard) + run.opened(functools.partial(self._opened, agents, watching, tally)) + self._draw() def drive() -> int: + from hmz.flows import BudgetExceeded, FlowException + from hmz.runtime import Refused + try: - runner.run(task) + run.run() + except asyncio.CancelledError: + pass # stopped by hand, which said so as it was stopped + except Refused as why: + self._on_screen(self.show, f"hmz: {why}", "red") + except BudgetExceeded as why: + # The ordinary end of a budgeted loop rather than a crash: what a run is + # given a budget for. + self._on_screen(self.show, f"hmz: stopped -- {why}", "yellow") + except FlowException as why: + # A failure the flow API has a name for -- a harness that would not take the + # turn, a machine that went away, a flow that refused what it was handed -- + # which is said as what it is rather than as a traceback nobody can act on. + self._on_screen(self.show, f"hmz: {type(why).__name__}: {why}", "red") finally: tally.stops() # read once more, for what the last turn wrote on its way out watching.stops() # the clock the rate is over is the run's, and it is over # Only this run's own, and only while it is still the one running. A flow - # takes a while to unwind after it is stopped -- a loop sleeps off its round, - # a server is given seconds to go -- and the next flow may have started in - # the meantime. Clearing then would leave the running one unreachable, and - # saying it was done would be saying it of the wrong flow. - if self._stopping is agents: + # takes a while to unwind after it is stopped, and the next flow may have + # started in the meantime. Clearing then would leave the running one + # unreachable, and saying it was done would be saying it of the wrong flow. + if self._stopping is run: # Stopped by hand, and now finished unwinding: there is nothing left for # the press that does not wait for it to reach. - self._stopping = [] - if self._agents is agents: - self._agents = [] - with contextlib.suppress(RuntimeError): - self.call_from_thread( - self.show, "[dim]— the flow is done —[/dim]" - ) + self._stopping = None + if self._run is run: + self._run, self._agents = None, [] + self._spoke.set() + self._answered.set() + self._on_screen(self.show, "[dim]— the flow is done —[/dim]") # And whatever it never got round to taking, which is now on its way # nowhere: a flow that ends of its own accord strands the pin exactly as # one that is stopped does. @@ -3408,6 +3306,40 @@ def drive() -> int: self._background(drive) + def _opened( + self, + agents: list[AgentBase], + monitor: Monitor, + tally: Tally, + role: str, + agent: AgentBase, + session: SessionBase, + ) -> None: + """Takes one session a run has just opened as one of the run's own. + + Told on the run's own thread, before the session's first turn, with the agent behind + it already named for its role: its transcript is that role's, what its backend counts + is said to the monitor, and whichever turn of it starts next takes the oldest line + that was held. + + Args: + agents: The run's agents, which this one joins. + monitor: The run's monitor. + tally: What reads the run's logs. + role: The role it was opened for. + agent: The agent behind it. + session: Its conversation. + """ + del role, session + agent.waiting = self._at_turn_start + agents.append(agent) + # What its backend counts, said before its first turn: a kind nothing was spent on + # this turn is missing from that turn's reckoning exactly as a kind the CLI never + # counts is, and what is drawn of a run driving two backends has to tell the two + # apart to say which of its figures are whole. + monitor.reporting(agent.id, type(agent).counts) + tally.add(agent) + def _remember_btw(self, agent: AgentBase, event: Event) -> None: """Keeps a compact progress record for future side questions. @@ -3418,7 +3350,7 @@ def _remember_btw(self, agent: AgentBase, event: Event) -> None: # A stopped flow can take a moment to unwind while a new one is already up. Its old # watcher is still bound to this method, but its events must not become progress for # the new run. - if self._agents and not any(agent is held for held in self._agents): + if self._run is not None and not any(agent is held for held in self._agents): return if event.kind not in { "begins", @@ -3502,6 +3434,12 @@ def _heard( # only when one arrives stands still through all of them. self._monitor.stirring() self._remember_btw(agent, event) + if event.kind == "result": + # Kept to tell a flow saying an agent's answer back to the person -- a + # conversation asking what next -- from a question it asks them. + self._last_answer = event.text + elif event.kind == "asks": + self._last_asked = event.text if event.kind == "took": # The agent saying a word put into its turn is now in front of it, which is the # one thing that makes a word said rather than posted. @@ -3620,19 +3558,20 @@ def _heard( packs=False, ) - @staticmethod - def _conversation(agent: AgentBase, session: SessionBase | None) -> str: - """Which of an agent's conversations a turn is being taken in, where it has several. + def _conversation(self, agent: AgentBase, session: SessionBase | None) -> str: + """Which of a role's conversations a turn is being taken in, where it has several. Args: - agent: Whose turn it is. + agent: Whose turn it is -- one of the agents behind the role's sessions. session: The conversation it is in, or None where the agent said it. Returns: - Which one, counting from one, and nothing at all for an agent holding one -- there + Which one, counting from one, and nothing at all for a role holding one -- there being nothing to tell it apart from. """ - held = agent.sessions + held = [ + one for each in self._of(agent.id) for one in each.sessions + ] or agent.sessions if session is None or len(held) < 2: # noqa: PLR2004 -- one is none to tell apart return "" at = next( @@ -3740,7 +3679,7 @@ def _said(self, text: str) -> None: self._said_by_you(text) self._answer = text self._answered.set() # and the turn waiting on it carries on - elif self._agents: + elif self._run is not None: self._interject(text) else: self._said_by_you(text) @@ -3762,30 +3701,87 @@ def _starts(self, task: str) -> None: telemetry.snag("nothing-started", because="no coding agent installed") self.show("hmz: no coding agent is installed here", "red") return - named = [part for runs in self._models for part in ("-a", runs.spec)] - self._flow(["-f", self._flow_named, *named, task]) + self._flow(task) - def _ask(self, agent: AgentBase, question: Question) -> str | None: - """Puts a question a turn stopped on to whoever is at this prompt, and waits for them. + def _outworlder_asks(self, generation: int, question: Question) -> str | None: + """Puts what a flow asks the person outside it to this prompt, and waits for them. - Called from the turn's own thread, which is the one that waits: the agent has stopped - working until this is answered. `/afk` is what says nobody is here to answer, and so - is a flow that ends or is stopped while the question is still up -- neither leaves a - turn waiting on a reply that is not coming. - - Asked of the agent that is asking rather than of whatever is running now, as - :meth:`_listen` is, so that a flow on its way out cannot take the answer meant for - the flow that replaced it. + Called from a thread of the run's own, which waits here. A question asked while a + turn is open -- an agent that stopped to ask, put to the person by its flow -- or one + with answers to choose from is shown and answered by the line typed after it, since a + line typed into an open turn would otherwise go to the turn. Anything else is what to + say next, answered by the next line typed, a line typed before it was asked included; + what was asked is shown first, unless it is what an agent has just answered, which the + transcript already shows -- a conversation saying back what was said. `/afk`, a run + that has ended or been stopped, and an interface that has gone each answer nobody, + which the flow hears as whoever is outside the run being away. Args: - agent: Whose turn stopped to ask. - question: What the agent wants to know. + generation: Which run is asking, so that a run on its way out cannot take the + answer meant for the run that replaced it. + question: What it asks. Returns: What was typed, or None if nobody was there to type it. """ - if self._afk or agent.stopped or agent not in self._agents: + if not self._live(generation): return None + if question.options or len(self._working): + return self._ask(generation, question) + said = question.text.strip() + if said and said != self._last_answer.strip(): + with contextlib.suppress(RuntimeError): # or the interface has gone + self.call_from_thread(self._show_question, question) + return self._listen(generation) + + def _live(self, generation: int) -> bool: + """Whether the run asking is still the one going, and somebody is here to answer.""" + return ( + not self._afk and generation == self._generation and self._run is not None + ) + + def _listen(self, generation: int) -> str | None: + """Waits at the prompt for a flow that has nothing to do until it is told something. + + Nothing on the event loop is touched, so the interface goes on being an interface + while a flow waits in it. + + Args: + generation: Which run is waiting. + + Returns: + What was said next, or None once this flow is over -- stopped by hand, or the + interface going away, either of which has to release this rather than leave a + thread waiting on a prompt that is not there. + """ + self._awaiting = True + try: + while True: + # Cleared before the queue is read, so that a line arriving between the two + # sets it again and is not waited through. + self._spoke.clear() + if not self._live(generation): + return None + if held := self._take(): + # Whatever turn this answer starts is that line's turn, and takes + # nothing else out of the queue on the way in. + with self._saying: + self._handed = True + return "\n\n".join(held) + self._spoke.wait(_REFRESH) + finally: + self._awaiting = False + + def _ask(self, generation: int, question: Question) -> str | None: + """Puts a question the flow asks to whoever is at this prompt, and waits for them. + + Args: + generation: Which run is asking. + question: What it wants to know. + + Returns: + What was typed, or None if nobody was there to type it. + """ # Cleared before the question goes up, so that an answer arriving between the two is # not cleared away with it. self._answered.clear() @@ -3794,38 +3790,44 @@ def _ask(self, agent: AgentBase, question: Question) -> str | None: self.call_from_thread(self._show_question, question) while not self._answered.wait(_REFRESH): # `/afk` while the question is up says so too, or saying you are away would - # leave the turn waiting on the answer you had just declined to give. - if self._afk or agent.stopped or agent not in self._agents: + # leave the flow waiting on the answer you had just declined to give. + if not self._live(generation): break self._asking = None return self._answer or None def _show_question(self, question: Question) -> None: - """Shows what a question offers, under the question itself. + """Shows a question the flow asks, and what it will take for an answer. - The question is shown as the turn says it, like anything else the agent said. What is - added here is what it will take for an answer, which only the one asking knows -- and - it goes against the agent the question went against, or the two would be read apart. + On whichever transcript is being read, the question being the flow's rather than any + one agent's -- unless an agent has just stopped to ask the same thing, which is + already on its own transcript and is not said twice. Args: - question: What the agent wants to know. + question: What the flow wants to know. """ - asked = self._asked_on + asked = question.text.strip() + repeated = bool(self._last_asked) and asked == self._last_asked.strip() + self._asked_on = None if not repeated else self._asked_on + if not repeated: + self._part(None, f"[yellow]{_SAID}[/] {escape(asked)}", packs=False) for option in question.options: - self._into(asked, f" [dim]· {escape(option)}[/dim]") - self._into(asked, " [dim]type an answer, or /afk to stop being asked[/dim]") + self._into(self._asked_on, f" [dim]· {escape(option)}[/dim]") + self._into( + self._asked_on, " [dim]type an answer, or /afk to stop being asked[/dim]" + ) @property def _set_up(self) -> bool: - """Whether there is something for each of the flow's agents to run on. + """Whether there is an agent for each of the flow's agent roles. There is always a flow -- the interface opens on one -- so this is only ever short of - an agent, which is a machine with no coding agent installed on it. A flow that asks - for none is not short of anything: the person at this prompt is an agent it is handed - rather than one anybody chooses, so a flow that talks only to them has everything it - needs the moment it is chosen. + an agent, which is a machine with no coding agent installed on it. A flow with no + agent role is not short of anything: whoever is outside it is at this prompt. """ - return bool(self._models) or not self._wanted + return all( + role in self._models and self._models[role].spec for role in self._named_by + ) def _interject(self, text: str) -> None: """Puts something in the queue for the flow, and sends it if nothing is in the way. diff --git a/src/hmz/tui/pick.py b/src/hmz/tui/pick.py index 3cc04fd9..7bcf2515 100644 --- a/src/hmz/tui/pick.py +++ b/src/hmz/tui/pick.py @@ -16,13 +16,13 @@ read as one more thing to pick, and a menu whose way out looks like one of its answers is a menu nobody can see the way out of. -One agent is three steps, in this order and one agent at a time: which coding agent takes its -turns and which account it runs as (:class:`RunsAs`), which model it runs and at what effort -(:class:`Models`), and -- only where the flow said that one may be pointed at a machine -- -where its work lands (:class:`Anchors`). The order is the order of what depends on what: an -account belongs to a backend and a model belongs to the CLI that runs it, so neither can be -asked before the CLI has been. The backends are read one at a time, a tab apiece: the ones -installed here plus an optional one the sheet can teach somebody to install. Every model of +A flow is set up by role: each agent role it declares is a CLI, an account, a model and an +effort (:class:`Agent`); each environment role is where it is, written as `-e` writes it; and +beside them are the flow's own params and what a run of it may spend. The order of one agent's +rows is the order of what depends on what: an account belongs to a backend and a model belongs +to the CLI that runs it, so neither can be asked before the CLI has been. The backends are read +one at a time, a tab apiece: the ones installed here plus an optional one the sheet can teach +somebody to install. Every model of every CLI in one list is a list that grows each time any of them ships a model. The effort is the line with the arrows on it, exactly as Claude Code's is, and beside it the things that really are side questions about the same agent. @@ -53,7 +53,7 @@ runtime_checkable, ) -from pydantic import BaseModel, Field +from pydantic import BaseModel, Field, field_validator, model_validator from rich.markup import escape from textual import events, on, work from textual.await_complete import AwaitComplete @@ -64,30 +64,24 @@ from textual.widgets.option_list import Option from hmz.coganchor import backends -from hmz.coganchor.agents import ANYONE, FLOW, SWARM, USER, anchored, driver -from hmz.coganchor.agents.allowance import ( - Allowance, - allowed, - blinded, - unreadable, - unwatched, -) +from hmz.coganchor.agents import ANYONE, FLOW, SWARM, USER, driver from hmz.coganchor.prices import money +from hmz.flows import Budget from hmz.runtime import telemetry from hmz.runtime.kept import Runs from hmz.runtime.telemetry import KEPT, SAYS, SENT -from .discover import installed, machines, ready_to_open +from .discover import installed, ready_to_open from .monitor import Counted, Shape, lasting, short, thousands from .selecting import Choices if TYPE_CHECKING: - from collections.abc import Callable, Generator, Iterable, Mapping, Sequence + from collections.abc import Callable, Generator, Mapping, Sequence from pydantic.fields import FieldInfo from textual.app import App, ComposeResult - from hmz.coganchor.agents import AgentBase, Board, Moment + from hmz.coganchor.agents import AgentBase, Board from hmz.coganchor.backends import Model, Way # Under another name, because `Falls` here is the sheet one account's chain is chosen on @@ -97,7 +91,7 @@ from hmz.coganchor.providers import Provider from hmz.daemon import Hmz from hmz.runtime.epic import Ran - from hmz.runtime.flowing import Flowverse, Offer, Place + from hmz.runtime.flowing import AgentRole, EnvRole, Flowverse, Offer from .monitor import Monitor, Under @@ -112,13 +106,13 @@ "Accounts", "Agent", "Alike", - "Anchors", "Backends", "Catalogue", "Chosen", "Clis", "Configures", "Confirms", + "Declared", "Does", "Doing", "Drafts", @@ -141,66 +135,35 @@ "Signs", "Speaks", "Ways", + "budget_of", "called", - "config_of", - "model_of", + "declared_of", + "named_as", "opens_on", - "places_of", - "pointed", + "params_model", + "params_of", "reads", + "serves", "setting", "settled", + "spent", ] -def called(places: tuple[Place, ...], at: int) -> str: +def called(roles: Sequence[str], at: int) -> str: """What to call the agent being configured, which every step of configuring it says. In one place because it is said in three, and an agent that read as two different things between one step and the next would be two. Args: - places: One place per agent the flow drives, in the order it takes them. + roles: The agent roles the flow declares, in the order it declares them. at: Which of them is being asked about, counting from zero. Returns: - The name the flow calls it, or where it comes among them for a flow that named none. - """ - return places[at].name or f"agent {at + 1} of {len(places)}" - - -def pointed(place: Place) -> bool: - """Whether where one agent works is a question anybody is asked about it. - - Only for a place the flow declared `Remote`: a flow that says so is a flow that expects - to be told where that agent works, and one that says nothing has said its agent works - here. A container the flow named is not asked about either -- the flow settled it, and - nobody else has any say in it. - - Args: - place: What the flow declared. - - Returns: - True if there is a machine to be chosen for it, which is a step of its own. + The role, which is what the flow calls it. """ - from hmz.coganchor.agents import Remote - - return place.where is Remote or isinstance(place.where, Remote) - - -def _settled(place: Place) -> str: - """The container a flow put one of its agents in, where it named one. - - Args: - place: What the flow declared. - - Returns: - The image, or "" for an agent that works here and one that is asked where it works -- - neither of which is something the flow settled. - """ - from hmz.coganchor.agents import Isolated - - return place.where.image if isinstance(place.where, Isolated) else "" + return roles[at] if at < len(roles) else f"agent {at + 1} of {len(roles)}" #: What Claude Code rules the top of a sheet with, and how far in everything under it sits. @@ -330,9 +293,9 @@ def _holds(held: Held) -> str: def reads( - named: tuple[str, ...], runs: list[Runs], holding: Sequence[Held] = () + named: tuple[str, ...], runs: Sequence[Runs], holding: Sequence[Held] = () ) -> list[str]: - """One line per agent a flow drives: what it runs, where, and what it is holding. + """One line per agent role a flow declares: what it runs, and what it is holding. In one place because it is read in two -- above the prompt while a flow runs, and under the diagram on `/monitor` before any agent has worked -- and an agent that read as two @@ -341,8 +304,8 @@ def reads( line without it, and it says nothing there. Args: - named: What the flow calls each of them, "" apiece where it names none. - runs: What each of them runs, and where its turns land. + named: The role each of them fills. + runs: What each of them runs. holding: The conversations each of them has open, in the same order, or nothing at all for a flow that is not running -- which holds none. @@ -355,19 +318,9 @@ def reads( for part in ( named[at] if at < len(named) else "", one.spec, - one.anchor, - # A word either way, because both answers are worth reading: the rung this - # agent was narrowed to, or the word for not having been narrowed at all -- - # which leaves it at what it was configured with, and which a gap on the - # line could not tell from a setting that had gone missing. The account it - # runs as is not like it: one that says nothing is the one this machine is - # already signed in as, which is the line saying nothing new. - backends.permitted(one.permission), + # The account it runs as, where it is not the one this machine is already + # signed in as -- which is the line saying nothing new. one.provider, - # And off is the only one of the three worth a word: on is what a CLI that - # searches does anyway, and an agent nobody was asked about is one this line - # has nothing to say about either. - "no web search" if one.web_search is False else "", _holds(holding[at]) if at < len(holding) else "", ) if part @@ -377,7 +330,7 @@ def reads( _SHEET = """ -Anchors, Backends, Configures, Flows, Models, Monitoring, Providers, RunsAs, Signing, +Backends, Configures, Flows, Models, Monitoring, Providers, RunsAs, Signing, Ways { align: center middle; background: $background; } #sheet { width: 100%; height: auto; padding: 0; } @@ -1171,39 +1124,134 @@ async def asks_to_save(self) -> None: class Chosen(NamedTuple): """What the flow menu was answered with: what to run, on what, and set up how. - One answer rather than three, because the menu is one thing answered once: what is held + One answer rather than four, because the menu is one thing answered once: what is held on each of its pages lands together when it is saved, or none of it does. Attributes: flow: The flow to run, by the name it was offered under. - agents: What each of its agents is, in the order the flow takes them. - config: What the flow itself is set up with, or None for a flow that takes no setting - up and one that was left as it comes. - budget: What a run of it may spend, or None to run under whatever the flow itself - says -- which is what a flow nobody has set one for here does. Beside the config - rather than inside it, because it is a setting of the run: the flow declares at most - a default and never holds itself to one. + agents: What each of its agent roles runs, by role. + envs: Where each of its environment roles is, by role, as `-e` spells one. + params: What the flow itself is set up with, or None for a flow that takes no params + and one that was left at its defaults. + budget: What a run of it may spend, or None for none -- which only a flow humanize + ships may be run with. """ flow: str - agents: tuple[Runs, ...] - config: BaseModel | None = None - budget: Allowance | None = None + agents: dict[str, Runs] + envs: dict[str, str] = {} # noqa: RUF012 -- a NamedTuple's default, never written to + params: BaseModel | None = None + budget: Budget | None = None + + +class Declared(NamedTuple): + """What a flow declares that the menu asks about, read once per flow. + + Attributes: + agents: The agent roles somebody chooses an agent for, in the order the flow declares + them. An `Outworlder` role is whoever is at this prompt, and is not among them. + envs: The environment roles somebody names a place for. A `LocalEnv` role is the + workspace a run is started in, and is not among them either. + params: What the flow can be set up with. + unbounded: Whether a run of it needs no budget: a flow humanize ships -- `chat`, a + conversation, which stops when the person does. + resumable: Whether a run of it can be picked up where it left off. + """ + + agents: tuple[AgentRole, ...] + envs: tuple[EnvRole, ...] + params: type[BaseModel] + unbounded: bool = False + resumable: bool = False + + @property + def roles(self) -> tuple[str, ...]: + """The agent roles, by name, in the order the flow declares them.""" + return tuple(one.name for one in self.agents) + + @property + def places(self) -> tuple[str, ...]: + """The environment roles, by name, in the order the flow declares them.""" + return tuple(one.name for one in self.envs) + + +def declared_of(flow: str) -> Declared | None: + """What a flow declares, or None for a flow that will not load. + + Args: + flow: The flow, by the name it was offered under -- not by the file that name resolves + to, since a module may hold several and which of them was asked for is the half after + the colon. + + Returns: + Its roles, params and marks, and None where reading the flow raised at all -- which is + a flow to report rather than a reason for a menu not to draw. + """ + from hmz.runtime.flowing import builtin, resolved + + try: + # Loaded once, and asked both questions: loading a flow reads its directory. + impl = resolved(flow) + said = impl.describe() + unbounded = builtin(impl) + except Exception: # noqa: BLE001 -- a flow that will not load is still not a crash + return None + return Declared( + tuple(one for one in said.agents if not one.auto), + tuple(one for one in said.envs if not one.auto), + said.params, + unbounded=unbounded, + resumable=said.resumable, + ) + + +def _harness(backend: str) -> str: + """Which harness a CLI is, as the flow API names it: its own name, or `acp`.""" + from hmz.flows import HarnessKind + + try: + return HarnessKind(backend).value + except ValueError: + return HarnessKind.ACP.value + + +def serves(backend: str, role: AgentRole | None) -> bool: + """Whether one CLI could fill an agent role, as a run of the flow would ask it. + + Args: + backend: The CLI. + role: What the flow declared of the role, or None for an agent that is no flow's -- + either end of a fallback step, which a flow says nothing about. + + Returns: + False for a CLI that is not the harness the role names, or that cannot do what the + role declares it must; True otherwise. + """ + if role is None: + return True + from hmz.flows import HarnessKind + from hmz.runtime.flowing import HARNESS_CAPABILITIES + + harness = HarnessKind(_harness(backend)) + if role.harness is not None and harness != role.harness: + return False + return role.capabilities <= HARNESS_CAPABILITIES[harness] def opens_on( - agents: Mapping[str, tuple[Model, ...]], *, goals: bool = True + agents: Mapping[str, tuple[Model, ...]], role: AgentRole | None = None ) -> list[Runs]: - """The one agent to fall back on where nothing has been remembered for a place. + """The one agent to fall back on where nothing has been remembered for a role. - The first backend installed here that has said what it runs and can be opened without - further setup, at the first model it named -- which is that CLI's own idea of what it runs - by default, and the only idea of it worth having. Nothing is written down here: a model - named in this file would be a model this file was right about on the day it was written. + The first backend installed here that has said what it runs, can be opened without + further setup and could fill the role, at the first model it named -- which is that CLI's + own idea of what it runs by default, and the only idea of it worth having. Nothing is + written down here: a model named in this file would be a model this file was right about + on the day it was written. Args: agents: The backends there are, and what each of them says it runs. - goals: Whether backend goals start available to it. + role: What the flow declared of the role, or None for any. Returns: The one agent, or nothing at all where no backend here has both said what it runs and @@ -1211,7 +1259,7 @@ def opens_on( """ where = Path.cwd() for backend, found in agents.items(): - if found and ready_to_open(backend, where): + if found and serves(backend, role) and ready_to_open(backend, where): # Not the hardest effort, which is where the cursor starts: that is the one to # reach for, and this is the one to spend before anybody has asked for anything. # `high` where the model takes it, which is nearly always -- and the least it @@ -1223,36 +1271,16 @@ def opens_on( effort = "high" if "high" in one.efforts else "" if not effort and one.efforts: effort = one.efforts[-1] - return [ - Runs(f"{backend}/{one.name}:{backends.written(effort)}", goals=goals) - ] + return [Runs(f"{backend}/{one.name}:{backends.written(effort)}")] return [] -def places_of(flow: str) -> tuple[Place, ...] | None: - """The agents a flow drives, or None for a flow that will not load. - - Args: - flow: The flow, by the name it was offered under -- not by the file that name resolves - to, since a file may hold several and which of them was asked for is the half after - the colon. - - Returns: - One place per agent it drives, and None where reading the flow raised at all -- which - is a flow to report rather than a reason for a menu not to draw. - """ - try: - return _hmz().flows.places(flow) - except Exception: # noqa: BLE001 -- a flow that will not load is still not a crash - return None - - def why_not(flow: str) -> str: """Why a flow will not load, in the words of whatever refused it. "will not load" on its own is a dead end: the reasons are nothing alike -- a flowverse that has not been fetched, a module the flow imports that is not installed, a syntax - error somebody just wrote, a file holding several flows and none of them named -- and + error somebody just wrote, a module holding several flows and none of them named -- and each is fixed somewhere else. So the reason is read off the exception rather than swallowed, and shown where the flow is picked. @@ -1267,7 +1295,7 @@ def why_not(flow: str) -> str: that loads, this being asked only of one that did not. """ try: - _hmz().flows.places(flow) + _hmz().flows.declared(flow) except Exception as why: # noqa: BLE001 -- the reason is the answer here said = str(why).strip().splitlines() first = said[0].strip() if said else "" @@ -1329,23 +1357,24 @@ def _wont_load(flow: str, also: str = "") -> str: return bad(said) -def model_of(flow: str) -> type[BaseModel] | None: - """What a flow says it can be set up with, if it says anything. +def params_model(flow: str) -> type[BaseModel] | None: + """What a flow can be set up with, where it takes anything at all. Args: flow: The flow, by name or as a path. Returns: - The model to ask with, or None for a flow that takes no setting up -- and for one that - will not load, which is a flow to report where it is run rather than here. + Its params model, or None for a flow whose params have no fields -- a sheet with + nothing on it is not a question -- and for one that will not load, which is a flow to + report where it is run rather than here. """ - try: - return _hmz().flows.configures(flow) - except Exception: # noqa: BLE001 -- a flow that will not load is still not a crash + declared = declared_of(flow) + if declared is None or not declared.params.model_fields: return None + return declared.params -def config_of(flow: str, kept: dict[str, Any]) -> BaseModel | None: +def params_of(flow: str, kept: Mapping[str, Any]) -> BaseModel | None: """How a flow was last set up, read back through the flow's own model rather than trusted. Args: @@ -1353,127 +1382,116 @@ def config_of(flow: str, kept: dict[str, Any]) -> BaseModel | None: kept: What was written down for it, field by field. Returns: - What it was set up with, or None for a flow that takes no setting up, has not been set + What it was set up with, or None for a flow that takes no params, has not been set up here, or has since changed enough that what was kept no longer reads -- a settings file is a convenience, and one that no longer fits is one to start over from. """ - model = model_of(flow) + model = params_model(flow) if model is None or not kept: return None try: - return model.model_validate(kept) + return model.model_validate(dict(kept)) except Exception: # noqa: BLE001 -- what was kept no longer fits the flow return None -def dimensions(said: Allowance) -> dict[str, float]: - """An allowance as the three fields a sheet asks for and a settings file writes down.""" - return {"hours": said.hours, "tokens": said.tokens, "dollars": said.dollars} +def spent(budget: Budget) -> str: + """What a budget caps, shortest first, as a row says it. + Args: + budget: The budget. -def _spending(held: Allowance | None, declared: Allowance | None) -> str: + Returns: + Each limit it sets -- the time, the output tokens, the money -- and `no limit` for the + one a conversation runs under, whose one cap is an infinite cost. + """ + import math + + caps: list[str] = [] + if budget.duration is not None: + caps.append(lasting(budget.duration.total_seconds())) + if budget.output_tokens is not None: + caps.append(f"{thousands(budget.output_tokens)} out") + if budget.cost is not None and not math.isinf(budget.cost): + caps.append(money(budget.cost)) + if not caps: + return "no limit" + return ", ".join(caps) + ("" if budget.graceful else ", cut mid-turn") + + +def _spending(held: Budget | None, *, unbounded: bool) -> str: """What a run of this flow may spend, said the way a row about it says it. - Said on the row rather than only inside the sheet it opens, because an allowance nobody - can see without opening something is one nobody checks: the row is where a person finds - out that the run they are about to start has no cap on it. + Said on the row rather than only inside the sheet it opens, because a budget nobody can + see without opening something is one nobody checks: the row is where a person finds out + what the run they are about to start is held to -- or that it is held to nothing yet, + which is a run that will not start. Args: - held: What was set here, or None for a flow nobody has set one for. - declared: What the flow itself says, or None for a flow with no opinion. + held: What was set here, or None for none. + unbounded: Whether the flow needs none: a conversation, which stops when you do. Returns: - The caps, shortest first, or a line saying there are none. + The caps, or what having none comes to. """ - said = held if held is not None else declared - if said is None or not said.bounded: - return "nothing stops this run" - caps = [ - f"{said.hours:g}h" if said.hours else "", - f"{said.tokens:g}M out" if said.tokens else "", - money(said.dollars) if said.dollars else "", - ] - whose = "" if held is not None else ", as the flow has it" - return f"stops at {', '.join(one for one in caps if one)}{whose}" + if held is None: + return ( + "none needed; it stops when you stop talking" + if unbounded + else "none yet; a run is given one" + ) + return f"stops at {spent(held)}" -def budget_of(flow: str) -> Allowance | None: +def budget_of(flow: str) -> Budget | None: """What a run of one flow here was last set to be allowed to spend. Args: flow: The flow. Returns: - The allowance, or None for a flow nobody has set one for here -- which is a run under - whatever the flow itself declares. What was written down is read back rather than - trusted, so a settings file somebody edited by hand into something that is not an - allowance is one the flow's own default is used instead of. + The budget, or None for a flow nobody has set one for here. What was written down is + read back rather than trusted, so a settings file somebody edited by hand into + something that is not a budget is one that asks again. """ - from hmz.coganchor.agents.allowance import written - kept = _hmz().settings.budget(flow) if not kept: return None try: - return written(kept) + return Budget.model_validate(kept) except ValueError: return None -def declared_by(flow: str) -> Allowance | None: - """What the flow itself says a run of it is worth, if it says anything. - - Args: - flow: The flow, by name or as a path. - - Returns: - What it declared, `Allowance()` for a flow that says it is meant to run under nothing - at all, and None for a flow with no opinion -- which is the one the menu asks about. - """ - try: - return _hmz().flows.declared(flow) - except Exception: # noqa: BLE001 -- a flow that will not load is still not a crash - return None - - def settled( - runs: Sequence[Runs], - places: Sequence[Place], + runs: Mapping[str, Runs], + roles: Sequence[AgentRole], agents: Mapping[str, tuple[Model, ...]] | None = None, -) -> list[Runs]: - """One agent per place a flow drives, out of however many were remembered for it. +) -> dict[str, Runs]: + """One agent per role a flow declares, out of whatever was remembered for it. - A flow that has grown an agent since it was last run here is a flow with a place nothing - was remembered for, and one that has lost one is a flow with an agent nobody will drive. + A flow that has grown a role since it was last run here is a flow with a role nothing was + remembered for, and one that has lost one is a flow with an agent nobody will drive. Neither is a reason to start over: what is there is kept, and what is missing falls back - on the agent the interface opens talking to. + on the agent the interface opens talking to, where one here could fill that role. Args: - runs: What was remembered, in the order the flow took them then. - places: What the flow drives now. - agents: The backends there are, for the place nothing was remembered for, or None - where there is nothing to fall back on -- which leaves such a place unanswered. + runs: What was remembered, by role. + roles: What the flow declares now. + agents: The backends there are, for a role nothing was remembered for, or None where + there is nothing to fall back on -- which leaves such a role unanswered. Returns: - One apiece, with goals forced on for a place the flow declared it needs them at -- that - one is the flow's own requirement rather than anybody's choice. + One agent per role that has one, by role, in the order the flow declares them. """ - spare = opens_on(agents) if agents is not None else [] - held: list[Runs] = [] - for at, place in enumerate(places): - if at < len(runs): - one = runs[at] - elif spare: - # What the flow suggested for a place nothing was remembered for: a flow that - # says its agent starts without goals is one whose fallback agent starts that - # way too, rather than one whose suggestion only counts on a command line. - one = spare[0]._replace(goals=place.goals) - else: - # Nothing remembered and nothing to fall back on, which is a machine with no - # coding agent installed on it: a place with no agent is a place with no agent, - # and an agent naming no model would be a worse answer than none. - break - held.append(one._replace(goals=True) if place.goal else one) + held: dict[str, Runs] = {} + for role in roles: + one = runs.get(role.name) + if one is None and agents is not None: + spare = opens_on(agents, role) + one = spare[0] if spare else None + if one is not None: + held[role.name] = one return held @@ -1504,8 +1522,8 @@ def _cli(runs: Runs) -> str: def _model(runs: Runs) -> str: """What one agent of the menu runs, out of the `cli/model:effort` it was set up as. - Read from both ends, as :func:`hmz.runtime.kept.written` reads the same word: a model may - hold slashes of its own, while a CLI and an effort never do. + Read from both ends, as :func:`hmz.runtime.kept.read_back` reads the same word: a model + may hold slashes of its own, while a CLI and an effort never do. Args: runs: The agent. @@ -1516,45 +1534,25 @@ def _model(runs: Runs) -> str: return runs.spec.partition("/")[2].rpartition(":")[0] -def _counts(runs: Runs) -> bool: - """Whether the backend one agent of the menu is driven by reports what it writes. +def placed(role: str, spec: str) -> str: + """What is wrong with where an environment role was said to be, or "" for nothing. - Asked of the CLI rather than of the model, `counts` being that backend's own word for - what it can report: a CLI somebody added by hand is driven over a protocol that counts - nothing at all, so a cap in tokens on one of those is a cap that will never bite. + Read the way `-e` is read, so that what the menu takes is what a command line would. Args: - runs: The agent. + role: The role. + spec: Where it is, as `-e` spells one after `=`. Returns: - Whether a token cap could be read off it. True for a CLI nothing here drives, which is - an agent no run can be started with either -- there is nothing to warn anybody about. + Why it is not one, in words, or "" for a spec that reads. """ - try: - return "output" in driver(_cli(runs))[0].counts - except KeyError: - return True - + from hmz.runtime.flowing import SpecError, parse_envs -def _cannot_read(blind: Iterable[str]) -> str: - """The line under the question, for a run whose caps are set and cannot be read. - - :func:`hmz.coganchor.agents.allowance.unreadable` already names the dimension and why - nothing can read it, and says it in the line `hmz exec` prints on its way past. Said the - same way here: two wordings of one fact are two things to keep in step, and a person who - reads it in both places is reading about the same run. - - Args: - blind: The dimensions nothing can read, as `Reading.blind` names them. - - Returns: - The line, or "" for a run every cap of which can be read -- which is the ordinary - unbounded one, capped on nothing at all. - """ - said = unreadable(blind) - if not said: - return "" - return f"{said[:1].upper()}{said[1:]}, so that cap cannot stop this run." + try: + parse_envs([f"{role}={spec}"]) + except SpecError as why: + return str(why) + return "" #: What separates the two halves of a row's id among the flows: which place it came from, @@ -1600,10 +1598,11 @@ class Flows(Drafts[Chosen]): list of flows is being read; what can happen to a flowverse is the menu `v` opens, which is a question about the places rather than about which flow to run. - Choosing a flow asks what that flow itself takes, where it takes anything, and then opens - what will drive it. A key that set the flow up was a key nobody pressed: a flow with - settings is chosen in order to be run with settings, and the moment it is chosen is the - one moment somebody is thinking about that flow rather than about its agents. + Choosing a flow asks what that flow itself takes -- its params -- where it takes anything, + and then opens its roles: a row per agent role somebody chooses an agent for, a row per + environment role somebody says the place of, what a run of it may spend, and saving. The + roles the runtime fills -- whoever is at this prompt, the workspace a run starts in -- are + not rows: nobody chooses them. Nothing is applied by walking in or back out. What the menu holds is a draft of the whole of it, and it lands together from the save row or when saving is confirmed on the way out. @@ -1628,12 +1627,13 @@ class Flows(Drafts[Chosen]): def __init__( self, flow: str, - runs: Sequence[Runs], - config: BaseModel | None, + runs: Mapping[str, Runs], + params: BaseModel | None, agents: dict[str, tuple[Model, ...]], kept: dict[str, Any], *, - budget: Allowance | None = None, + envs: Mapping[str, str] | None = None, + budget: Budget | None = None, unavailable: frozenset[str] = frozenset(), running: bool = False, inside: bool = False, @@ -1642,16 +1642,16 @@ def __init__( Args: flow: The flow running now, or the one this workspace is set up to run. - runs: What each of its agents is, in the order the flow takes them. - config: What the flow itself is set up with, for one that takes setting up. + runs: What each of its agent roles runs, by role. + params: What the flow itself is set up with, for one that takes params. agents: The backends offered here, and what each of them says it runs. - budget: What a run of it may spend here, or None for a flow nobody has set one - for -- which runs under whatever the flow itself says. kept: What each flow was last set up with here, by flow -- read when the draft flow changes, so that turning to a flow this workspace has run finds it as it was left. + envs: Where each of its environment roles is, by role. + budget: What a run of it may spend here, or None for none yet. unavailable: The optional backends among them that still need installing. running: Whether a flow is running, which is what takes the flows away. - inside: Whether to open inside the flow's agents rather than on the flows, for a + inside: Whether to open inside the flow's roles rather than on the flows, for a menu opened already naming one -- a flow that was named has been chosen, so what is left to answer is what drives it. """ @@ -1659,32 +1659,28 @@ def __init__( self._agents = dict(agents) self._unavailable = unavailable self._kept = kept - # Said outright, both of them: the flow is read where it is set, so what it is has to + # Said outright, all of them: the flow is read where it is set, so what it is has to # be settled without reading what reads it. self._flow: str = flow - self._places: tuple[Place, ...] = places_of(flow) or () + #: What the flow declares, read once per flow rather than on every redraw: reading + #: it means importing the flow, and the roles page is drawn on every keystroke. + self._declared: Declared | None = declared_of(flow) + self._runs: dict[str, Runs] + self._envs: dict[str, str] if runs: - self._runs = ( - self._fitted(settled(runs, self._places, self._agents)) - if self._places - else list(runs) - ) - self._config = config + self._runs = self._fitted(dict(runs)) + self._envs = dict(envs or {}) + self._params = params self._budget = budget else: # A flow the interface is not set up on, opened straight into: what it was last # set up with here is what it opens holding, exactly as turning to it would be. - self._runs = self._fitted( - settled(self._remembered(flow), self._places, self._agents) - ) - self._config = config_of(flow, self._held(flow).get("config") or {}) + self._runs = self._fitted(self._remembered(flow)) + self._envs = self._placed(flow) + self._params = params_of(flow, self._held(flow).get("params") or {}) self._budget = budget_of(flow) - #: What the flow itself says a run of it is worth, read once per flow rather than on - #: every redraw: reading it means running the flow's own file, and the agents page is - #: drawn again on every keystroke. - self._declared = declared_by(self._flow) #: Every flow there is, read once: this is redrawn on every keystroke, and reading it - #: means running each flow file to see what it holds. Cleared when a flowverse is + #: means importing each flow to see what it holds. Cleared when a flowverse is #: fetched or taken away, which is when the list is something else. self._offers: list[Offer] | None = None #: Which row of the flows the cursor is on, as `where it came from` and `which flow`: @@ -1693,21 +1689,21 @@ def __init__( self._was = "" #: Which place's flows are being read, the arrows stepping between them. "" until the #: flows are first drawn: which place the flow in force came from is a thing only the - #: list of every flow there is can say, and reading that list is running every file. + #: list of every flow there is can say, and reading that list is importing every flow. self._where = "" #: What became of the last fetch, said under the list. self._said = "" #: What is being fetched now, so that a second fetch is not started over it and so #: that what is said under the list is what is being fetched. "" for none. self._fetching = "" - #: Whether the agents are the whole of this menu, there being no flows behind them - #: to step back to: while a flow runs choosing one is not offered, and a flow that was + #: Whether the roles are the whole of this menu, there being no flows behind them to + #: step back to: while a flow runs choosing one is not offered, and a flow that was #: named was chosen on the line that named it rather than picked out of a list. Esc #: reads off this -- a step back to a list nobody walked through is a step somebody #: did not take, and on a `$` that named a flow it would swallow the line typed with #: it. self._only = running or inside - #: Whether what is open is the agents of the flow rather than the flows. + #: Whether what is open is the roles of the flow rather than the flows. self._inside = self._only def searching(self) -> str: @@ -1765,50 +1761,63 @@ def _follows(self, listing: OptionList) -> None: if _HALVES in named: self._was = named - def _fitted(self, runs: Sequence[Runs]) -> list[Runs]: - """One row per agent the flow drives, whatever there was to fill it with. + def _fitted(self, runs: Mapping[str, Runs]) -> dict[str, Runs]: + """One agent per agent role the flow declares, whatever there was to fill it with. - A place nothing was remembered for and nothing falls back on still has a row here: - this is where it is set up, and a place with no row is a place nobody can answer. What - such a row says is that it has not been answered yet. + A role nothing was remembered for and nothing falls back on still has a row: this is + where it is set up, and a role with no row is a role nobody can answer. What such a + row holds is an agent that names nothing, which says it has not been answered yet. Args: - runs: What there is, in the order the flow takes them. + runs: What there is, by role. Returns: - One apiece, padded with an agent that names nothing. + One apiece, by role, in the order the flow declares them. """ - return [ - runs[at] if at < len(runs) else Runs("") for at in range(len(self._places)) - ] + declared = self._declared + if declared is None: + return dict(runs) + held = settled(runs, declared.agents, self._agents) + return {role: held.get(role, Runs("")) for role in declared.roles} def _held(self, name: str) -> dict[str, Any]: """What one flow was last set up with here, which is nothing for one never run.""" held = self._kept.get(name) return cast("dict[str, Any]", held) if isinstance(held, dict) else {} - def _remembered(self, name: str) -> list[Runs]: - """What one flow's agents were last set up as here, in the order it takes them. + def _remembered(self, name: str) -> dict[str, Runs]: + """What one flow's agent roles were last set up as here, by role. Args: name: The flow. Returns: - One apiece, and nothing at all for a flow this workspace has never run -- which is - a flow whose agents fall back on the one the interface opens talking to. + One agent per role that has one, and nothing at all for a flow this workspace has + never run -- which is a flow whose agents fall back on the one the interface opens + talking to. """ from hmz.runtime.kept import read_back - agents: dict[str, Any] = self._held(name).get("agents") or {} - return [ - runs - for runs in ( - read_back(cast("dict[str, Any]", one)) - for one in agents.values() - if isinstance(one, dict) - ) - if runs is not None - ] + agents = self._held(name).get("agents") + if not isinstance(agents, dict): + return {} + held: dict[str, Runs] = {} + for role, said in cast("dict[str, Any]", agents).items(): + runs = read_back(said) + if runs is not None: + held[str(role)] = runs + return held + + def _placed(self, name: str) -> dict[str, str]: + """Where one flow's environment roles were last said to be here, by role.""" + envs = self._held(name).get("envs") + if not isinstance(envs, dict): + return {} + return { + str(role): str(said) + for role, said in cast("dict[str, Any]", envs).items() + if isinstance(said, str) + } def _ask(self) -> None: """Puts up whichever of the two it opened on, and catches up on fetches.""" @@ -1860,15 +1869,14 @@ async def _catches_up(self) -> None: def reread(self) -> None: """Drops the flows read before a fetch landed, and draws the list again. - The places too, for the flow in force: its file may be one of the ones that just came - down, and the agents page is drawn off what was read from the old one. A flow that + What the flow in force declares too: its module may be one of the ones that just came + down, and the roles page is drawn off what was read from the old one. A flow that would not load before the fetch is exactly the flow this is for. """ self._offers = None if self._flow: - self._places = places_of(self._flow) or () - self._declared = declared_by(self._flow) - self._runs = self._fitted(settled(self._runs, self._places, self._agents)) + self._declared = declared_of(self._flow) + self._runs = self._fitted(self._runs) self._fill() def _walks(self, *, inside: bool) -> None: @@ -1896,8 +1904,9 @@ def _fill(self) -> None: # opened. self.query_one("#asked", Label).update(escape(self._flow)) self.query_one("#about", Label).update( - "What each agent it drives is: the CLI that takes its turns, the account " - "they run as, and the model at an effort." + "What each of its roles is given: an agent -- the CLI that takes its turns, " + "the account they run as, and the model at an effort -- or where an " + "environment is." ) self.tabbed("") self._agents_page() @@ -2104,76 +2113,102 @@ def _nothing(self) -> str: return "no flow of that name" return "" + def _roles(self) -> tuple[str, ...]: + """The agent roles somebody chooses an agent for, in the flow's own order.""" + return self._declared.roles if self._declared is not None else () + + def _places(self) -> tuple[str, ...]: + """The environment roles somebody names a place for, in the flow's own order.""" + return self._declared.places if self._declared is not None else () + def _agents_page(self) -> None: - """Puts up each agent the flow drives, followed by saving the complete setup.""" + """Puts up each role the flow declares, its budget, and saving the whole setup.""" listing = self.query_one("#choices", OptionList) - named = tuple(place.name for place in self._places) - lines = reads(named, self._runs) - # The save row is past the end of the numbering, so what is numbered is the agents. - self._counting = len(str(max(len(self._places), 1))) - # One row past the agents for what a run may spend, and one past that for saving. - at = min(listing.highlighted or 0, len(self._places) + 1) + roles, places = self._roles(), self._places() + runs = [self._runs.get(role, Runs("")) for role in roles] + lines = reads(roles, runs) + # The rows set apart are past the end of the numbering, so what is numbered is the + # roles: the agents, then the environments. + count = len(roles) + len(places) + self._counting = len(str(max(count, 1))) + # One row past the roles for what a run may spend, and one past that for saving. + at = min(listing.highlighted or 0, count + 1) rows = [ Option( self._row( seen, - called(self._places, seen), + called(roles, seen), lines[seen].split(_DOT, 1)[-1] - if self._runs[seen].spec + if runs[seen].spec else "not chosen yet", here=seen == at, inforce=False, ), id=f"={seen}", ) - for seen in range(len(self._places)) + for seen in range(len(roles)) ] + rows.extend( + Option( + self._row( + len(roles) + seen, + place, + self._envs.get(place) or "not said yet", + here=len(roles) + seen == at, + inforce=False, + ), + id=f"=@{place}", + ) + for seen, place in enumerate(places) + ) rows.append( Option( self._apart( "budget", - _spending(self._budget, self._declared), - here=at == len(self._places), + _spending( + self._budget, + unbounded=self._declared is not None + and self._declared.unbounded, + ), + here=at == count, ), id=f"={_BUDGET}", ) ) - rows.append( - self._saves("the flow and its agents", here=at == len(self._places) + 1) - ) + rows.append(self._saves("the flow and its roles", here=at == count + 1)) listing.set_options(rows) listing.highlighted = at self._drawn = listing.highlighted - said = self._said or ("" if self._places else self._noagents()) + said = self._said or ("" if count else self._noagents()) self.query_one("#tuning", Label).update( f"[$text-muted]{said}[/]" if said else "" ) # Esc is out of the menu only where there is no list of flows to step back to, # which is while a flow is running: the row says what the key does here. back = Key("esc", "close" if self._only else "back to the flows") - if at == len(self._places) + 1: + if at == count + 1: self._footed(Key("enter", "save"), back) else: - # What enter says is read off the row it is on -- `open` over an agent and `set` + # What enter says is read off the row it is on -- `open` over a role and `set` # over the budget, which `Sheet._footed` rewrites from the row set apart. self._footed(Key("enter", "open"), Key(_CHORD, "save"), back) def _noagents(self) -> str: - """Why there is no agent to set up, which is not always the same reason.""" - if places_of(self._flow) is None: + """Why there is no role to set up, which is not always the same reason.""" + if self._declared is None: return _wont_load(self._flow, "nothing here can be set up") - return f"{escape(self._flow)} drives no agents; it talks only to you" + return f"{escape(self._flow)} has no role to choose for; it talks only to you" @work async def _configures(self) -> None: - """Asks what the flow itself takes, and turns to what will drive it. + """Asks what the flow itself takes, and turns to its roles. - Which is the moment to ask it: a flow that takes settings has just been chosen, and + Which is the moment to ask it: a flow that takes params has just been chosen, and what it is set up with is a thing about the flow rather than about its agents. A flow that takes none is not asked -- a sheet with nothing on it is not a question -- and the walk is the same either way, so nobody has to know which kind they picked. """ - model = model_of(self._flow) + model = params_model(self._flow) if model is not None: showing = cast( "App[None]", @@ -2183,26 +2218,26 @@ async def _configures(self) -> None: Configures( self._flow, model, - self._config if isinstance(self._config, model) else None, + self._params if isinstance(self._params, model) else None, ) ) if held is not None: - self._config = held + self._params = held self.changed() # And walking out of it leaves the flow set up as the draft has it, which is - # still a flow to go on and answer the agents of. + # still a flow to go on and answer the roles of. self._walks(inside=True) @work async def _budgets(self) -> None: - """Asks what a run of this flow may spend, from the row on the agents page. + """Asks what a run of this flow may spend, from the row on the roles page. - A row reached rather than a sheet the walk goes through, because every flow has an - allowance and most runs want the one they already have: a page that had to be pressed - past on the way to the agents would be a question asked of somebody who has answered - it. It is on the agents page rather than among the flow's own settings because it is a - setting of the run: the flow's model would refuse the fields, and a budget read back - as one of the flow's settings is the one mistake this must not make. + A row reached rather than a sheet the walk goes through, because most runs want the + budget they already have: a page that had to be pressed past on the way to the roles + would be a question asked of somebody who has answered it. It is not among the + flow's params because it is a setting of the run: the flow's model would refuse the + fields, and a budget read back as one of the flow's params is the one mistake this + must not make. """ showing = cast( "App[None]", @@ -2212,16 +2247,63 @@ async def _budgets(self) -> None: Configures( self._flow, Budgeted, - Budgeted.model_validate(dimensions(self._budget)) - if self._budget is not None - else None, + Budgeted.of(self._budget) if self._budget is not None else None, asked=f"What a run of {self._flow} may spend", - about="Nothing is capped unless it is named, and 0 is no cap at all. " - "Whichever of them is reached first stops the run.", + about="A run stops at whichever limit it reaches first; at least one is " + "set. Empty or 0 is no limit on that one.", ) ) if isinstance(spends, Budgeted): - self._budget = Allowance(**spends.model_dump()) + self._budget = spends.budget() + self.changed() + self._fill() + + @work + async def _placing(self, role: str) -> None: + """Asks where one environment role is, as `-e` says it, and holds the answer. + + Args: + role: The environment role. + """ + from pydantic import create_model + + showing = cast( + "App[None]", + self.app, # pyright: ignore[reportUnknownMemberType] + ) + model = create_model( + "Where", + where=( + str, + Field( + default="", + description="local@/abs/path, or ssh@host/abs/path -- " + "ssh@host/~/path under the login's home", + ), + ), + ) + held = await showing.push_screen_wait( + Configures( + self._flow, + model, + model(where=self._envs.get(role, "")), + asked=f"Where {role} is", + about="The machine and the directory this environment role works in, " + "as -e says one after the role.", + ) + ) + if held is None: + return + said = str(held.model_dump().get("where") or "").strip() + wrong = placed(role, said) if said else "" + if wrong: + self._said = bad(escape(wrong)) + else: + if said: + self._envs[role] = said + else: + self._envs.pop(role, None) + self._said = "" self.changed() self._fill() @@ -2298,7 +2380,7 @@ async def action_verses(self) -> None: @on(OptionList.OptionSelected) def _took(self, event: OptionList.OptionSelected) -> None: - """Chooses the flow under the cursor, or opens the agent under it. + """Chooses the flow under the cursor, or opens the role under it. Args: event: What was chosen. @@ -2315,6 +2397,9 @@ def _took(self, event: OptionList.OptionSelected) -> None: if held == _BUDGET: self._budgets() return + if held.startswith("@"): + self._placing(held[1:]) + return try: at = int(held) except ValueError: @@ -2331,48 +2416,48 @@ def _chose(self, name: str) -> None: name: The flow, by the name it was offered under. """ if name != self._flow: - places = places_of(name) - if places is None: + declared = declared_of(name) + if declared is None: self._said = _wont_load(name) self._fill() return - self._flow, self._places = name, places - self._runs = self._fitted( - settled(self._remembered(name), places, self._agents) - ) - self._config = config_of(name, self._held(name).get("config") or {}) + self._flow, self._declared = name, declared + self._runs = self._fitted(self._remembered(name)) + self._envs = self._placed(name) + self._params = params_of(name, self._held(name).get("params") or {}) self._budget = budget_of(name) - self._declared = declared_by(name) self.changed() - # On to what the flow itself takes, where it takes anything, and then to what will - # drive it: three things about one flow, asked in the order they depend on nothing. + # On to what the flow itself takes, where it takes anything, and then to its roles: + # things about one flow, asked in the order they depend on nothing. self._configures() @work async def _configuring(self, at: int) -> None: - """Opens one agent of the flow, and holds whatever comes back as a draft. + """Opens one agent role of the flow, and holds whatever comes back as a draft. Args: at: Which of them, counting from zero. """ - if not 0 <= at < len(self._places): + declared = self._declared + if declared is None or not 0 <= at < len(declared.agents): return + role = declared.agents[at] showing = cast( "App[None]", self.app, # pyright: ignore[reportUnknownMemberType] ) chosen = await showing.push_screen_wait( Agent( - called(self._places, at), - self._runs[at], + role.name, + self._runs.get(role.name, Runs("")), self._agents, - place=self._places[at], + role=role, unavailable=self._unavailable, ) ) if chosen is None: return # walked out of it, which leaves that agent as the draft has it - self._runs[at] = chosen + self._runs[role.name] = chosen self.changed() self._fill() @@ -2390,63 +2475,52 @@ def leaving(self) -> None: super().leaving() def applied(self) -> None: - """Answers with the flow, its agents and how it is set up, all of it at once. + """Answers with the flow, its roles and how it is set up, all of it at once. - Unless one of them has not been answered: a flow driven by an agent that names no - model is a flow that stops on its first turn, and where it would be answered is what - to be looking at when that is said. + Unless something a run needs has not been answered: an agent that names no model is + a run that stops on its first turn, an environment nobody said the place of is a run + refused before it starts, and so is a run given no budget -- which only a flow + humanize ships may be. Each is said where it would be answered. """ + declared = self._declared missing = [ - called(self._places, at) - for at, one in enumerate(self._runs) - if not _complete(one) + role + for role, one in self._runs.items() + if not _complete(one) and (declared is None or role in declared.roles) ] + if declared is not None: + missing.extend( + one.name + for one in declared.envs + if one.required and not self._envs.get(one.name) + ) if missing: telemetry.snag("save-refused", missing=len(missing)) if not self._inside: - # Refused from the flows, on the way out: the agents are what is to be looked + # Refused from the flows, on the way out: the roles are what is to be looked # at, and the cursor was on a row of another list. self._walks(inside=True) - self._said = iffy(f"{escape(', '.join(missing))} has no model yet") + self._said = iffy(f"{escape(', '.join(missing))} is not set up yet") self._fill() return - # The one exit that makes an answer, so the one place to ask about a run nothing - # will stop: the save row and the question on the way out both come through here, - # and a check written at each of them is a check one of them would lose. - effective = allowed(self._budget, self._declared) - # And what those agents are is what says whether the caps can be read at all, which - # is asked here because here is where they have just been chosen: a cap in dollars on - # a model nobody prices, or in tokens on a CLI that counts none, is a cap that will - # never bite -- so a run held to nothing else is a run nothing will stop, and this is - # the last moment anybody is at a prompt to be told. - blind = blinded( - effective, - [_model(one) for one in self._runs], - counting=any(_counts(one) for one in self._runs), - ) - if unwatched(effective, self._declared, blind): - self._means_it(_cannot_read(blind)) + if self._budget is None and declared is not None and not declared.unbounded: + telemetry.snag("save-refused", missing=0) + if not self._inside: + self._walks(inside=True) + self._said = iffy( + "a run of this flow is given a budget: set what it may spend first" + ) + self._fill() return - self.dismiss(Chosen(self._flow, tuple(self._runs), self._config, self._budget)) - - @work - async def _means_it(self, about: str = "") -> None: - """Asks whether a run nothing will stop is what was meant, and saves if it is. - - Args: - about: What to say under the question, for a run whose caps are set and cannot be - read -- or "" for the ordinary one, which is a run capped on nothing at all. - """ - showing = cast( - "App[None]", - self.app, # pyright: ignore[reportUnknownMemberType] + self.dismiss( + Chosen( + self._flow, + dict(self._runs), + dict(self._envs), + self._params, + self._budget, + ) ) - if await showing.push_screen_wait(Unbounded(about)) != _KEEP: - telemetry.snag("unbounded-refused", flow=self._flow) - # Back to the menu holding everything it was holding, which is where a budget is - # set: the answer was "go and set one", and there is nothing else to do about it. - return - self.dismiss(Chosen(self._flow, tuple(self._runs), self._config, self._budget)) def _added(url: str, name: str) -> str: @@ -3160,102 +3234,6 @@ def action_done(self) -> None: self.dismiss(said) -class Anchors(Sheet[str]): - """Where one agent's turns land: this machine, or one an anchor reaches. - - A row of the sheet one agent is set up on, and only for a place the flow declared - `Remote`: a flow that says so is one that expects to be told where that agent works, and - one that said nothing has said its agent works here. - - The agent itself runs here whatever is chosen -- its credentials, its state directory and - its link to its model provider stay put. What moves is the project it reads and the - commands it runs, which is why this is a question about the agent rather than about the - flow: two agents of one flow may work on two machines. - - Listed rather than typed where the machine is one this one can see -- a container that is - running, a host with an entry in the ssh config -- and typed where it is not: a target is - a string, and the row for what has been typed appears among them, as soon as it reads as - one, while a search is running. - """ - - LETTERS: ClassVar = frozenset({"search"}) - - BINDINGS: ClassVar = [ - ("escape", "back", "back"), - Binding("s", "search", "search", priority=True), - ] - - def __init__(self, named: str, current: str = "") -> None: - """Initializes the moving. - - Args: - named: What the flow calls the agent this is about, which every step of configuring - it says. - current: The target this agent is on now, or "" for this machine. - """ - super().__init__() - self._named = named - self._current = current - self._found: list[tuple[str, str]] | None = None - - def _ask(self) -> None: - """Lists the machines there are to work on, and says what choosing one does.""" - self.query_one("#asked", Label).update(f"Select where {self._named} works") - self.query_one("#about", Label).update( - "The machine its work lands on. The agent runs here either way; what moves is " - "the project it reads and the commands it runs." - ) - self.query_one("#tuning", Label).update( - "[$text-muted]a target of your own -- ssh://HOST, docker://CONTAINER, " - "tcp://HOST:PORT -- joins these as it is typed into a search[/]" - ) - self._fill() - - def _fill(self) -> None: - """Puts the machines up, with whatever has been typed among them if it reads as one.""" - listing = self.query_one("#choices", OptionList) - if self._found is None: - # Once: looking costs a `docker ps`, and this is redrawn on every keystroke. - self._found = machines() - rows: list[tuple[str, str, str]] = [("", "this machine", "nothing moves")] - rows.extend((target, target, whose) for target, whose in self._found) - shown = [row for row in rows if self.fits(row[1], row[2])] - if self._typed and not any(row[0] == self._typed for row in shown): - # What has been typed, as soon as it is a target: a machine nobody here can see - # is still a machine, and this is the only way to name one. - try: - anchored(self._typed) - except ValueError: - pass - else: - shown.append((self._typed, self._typed, "as typed")) - self._counting = len(str(len(shown))) - at = min(listing.highlighted or 0, max(len(shown) - 1, 0)) - listing.set_options( - Option( - self._row( - seen, label, whose, here=seen == at, inforce=target == self._current - ), - # Every row is a target, and "" is this machine -- which an id of its own - # keeps tellable from a row that was never chosen. - id=f"={target}", - ) - for seen, (target, label, whose) in enumerate(shown) - ) - listing.highlighted = at if shown else None - self._drawn = at - self._footed(Key("enter", "choose"), Key("esc", "back")) - - @on(OptionList.OptionSelected) - def _took(self, event: OptionList.OptionSelected) -> None: - """Answers with the target that was picked. - - Args: - event: What was chosen. - """ - self.dismiss(str(event.option.id).removeprefix("=")) - - class Falls(Sheet[str]): """Which account a turn under this one carries on under when it fails. @@ -3531,31 +3509,72 @@ def _lasting(seconds: float) -> str: class Budgeted(BaseModel): """What a run of a flow may spend, as the menu asks it. - A model rather than three rows written by hand, so that the budget is asked with the same - sheet a flow's own settings are asked with: one place that knows how a number is typed, - stepped and read back, and three descriptions that say what each dimension means. What - comes out of it is fed to `Allowance`, which is where zero meaning "no limit" and a - negative meaning "correct this" are settled. + A model rather than four rows written by hand, so that the budget is asked with the same + sheet a flow's own params are asked with: one place that knows how a number is typed, + stepped and read back, and a description apiece saying what each limit means. What comes + out of it is a :class:`hmz.flows.Budget`, which is where "at least one limit" is settled. - Not `hmz.coganchor.agents.Budget`, which is a cap on one turn. This is the run. + Not a turn's budget, which is a flow's to give. This is the run's. """ - hours: float = Field( - default=0.0, - ge=0, - description="hours on the clock the whole run may take, 0 for as long as it takes", + duration: str = Field( + default="", + description="how long the run may take: 1h30m, 90s, PT2H; empty for no limit", + ) + cost: float = Field( + default=0.0, ge=0, description="US dollars it may cost, 0 for no limit" ) - tokens: float = Field( - default=0.0, - ge=0, - description="millions of output tokens it may come to, 0 for as many as it takes", + output_tokens: int = Field( + default=0, ge=0, description="output tokens it may come to, 0 for no limit" ) - dollars: float = Field( - default=0.0, - ge=0, - description="US dollars it may cost, 0 for whatever it costs", + graceful: bool = Field( + default=True, + description="off to cut a turn off mid-way when a limit is reached", ) + @field_validator("duration") + @classmethod + def _reads(cls, said: str) -> str: + """Refuses a duration that is not one, where it is typed.""" + from hmz.runtime.flowing import parse_duration + + if said.strip(): + parse_duration(said) + return said.strip() + + @model_validator(mode="after") + def _limits_something(self) -> Budgeted: + """Refuses a budget that limits nothing, which is no budget.""" + if not (self.duration or self.cost or self.output_tokens): + raise ValueError("set at least one of duration, cost and output_tokens") + return self + + @classmethod + def of(cls, budget: Budget) -> Budgeted: + """A budget, as the sheet shows it.""" + import math + + seconds = budget.duration.total_seconds() if budget.duration else 0.0 + return cls( + duration=f"{seconds:g}s" if seconds else "", + cost=budget.cost + if budget.cost is not None and not math.isinf(budget.cost) + else 0.0, + output_tokens=budget.output_tokens or 0, + graceful=budget.graceful, + ) + + def budget(self) -> Budget: + """What the sheet was answered with, as a run's budget.""" + from hmz.runtime.flowing import parse_duration + + return Budget( + duration=parse_duration(self.duration) if self.duration else None, + cost=self.cost or None, + output_tokens=self.output_tokens or None, + graceful=self.graceful, + ) + def _shown(value: object) -> str: """One setting's value, as a line about it says it. @@ -3592,6 +3611,20 @@ def _grouped(field: FieldInfo) -> str: return str(said) if said else "" +def named_as(ref: str) -> str: + """A flow's canonical ref as a line about it says it. + + Args: + ref: `:`, as the running tree names a call. + + Returns: + The module alone for the flow named after it -- `chat` rather than `chat:chat` -- and + the ref as it is otherwise. + """ + where, _, name = ref.partition(":") + return where if name == where else ref + + def _flowing(started: str) -> list[str]: """Which flow is running, and inside which, for the row that names one. @@ -3611,9 +3644,10 @@ def _flowing(started: str) -> list[str]: if not now: return [escape(started)] return [ - f"{' ' * at}{'▸ ' if at else ''}{escape(one.flow)}" + f"{' ' * (one.depth - 1)}{'▸ ' if one.depth > 1 else ''}" + f"{escape(named_as(one.ref))}" f" [$text-muted]{time.monotonic() - one.since:.0f}s[/]" - for at, one in enumerate(now) + for one in now ] @@ -4874,61 +4908,6 @@ def _fill(self) -> None: self._footed(Key("enter", "choose"), Key("esc", "back")) -class Unbounded(Popup): - """Whether a run nothing at all will stop is what was meant, asked as the menu is saved. - - Three caps and none of them set is a flow that will go until somebody notices -- for days, - and for whatever days of a model cost. That is a fair thing to ask for and a poor thing to - arrive at by not answering three questions, and the two look identical afterwards. So it - is asked once, here, where it can still be changed. - - Not asked of a flow that said so itself. A flow writing `@flow(budget=Allowance())` has - claimed in its own file that it is meant to run under nothing -- `chat` is a conversation - that ends when the person stops typing -- and a question asked every time somebody picks - one of those is a question nobody reads by the third time. - - The question and not a receipt: what is kept is written to a file that may not be - writable, and a box saying the run was saved would be claiming something this cannot - know. - """ - - #: The same box, said again for this class: every rule in this file selects by the name - #: of the sheet it is about, so a box drawn for another one is a rule of its own. - CSS = f"Unbounded {{ align: center middle; background: transparent; }}\n{_POPUP}" - - asked = "Nothing will stop this run." - - about = ( - "No hours, no output tokens and no dollars are capped, so it runs until it is " - "stopped by hand." - ) - - def __init__(self, about: str = "") -> None: - """Asks it, about this run. - - Args: - about: The line under the question, for a run whose caps are set and cannot be read - -- fifty dollars on a model nobody prices is a run with no limit on it, and the - box that said three caps were unset would be saying the one untrue thing about - it. "" for the ordinary one, which is a run capped on nothing at all. - """ - super().__init__() - if about: - self.about = about - - def rows(self) -> list[tuple[str, str, str]]: - """The two answers: mean it, or go back and cap something.""" - return [ - (_KEEP, "that is what I meant", ""), - (_DROP, "go back and set one", ""), - ] - - def _fill(self) -> None: - """Puts the two answers up, and says that esc is the second of them.""" - super()._fill() - self._footed(Key("enter", "choose"), Key("esc", "back")) - - #: What to do about a flow that is running when the interface is being closed: stop it, let #: go of the terminal and leave it running, or stay here after all. Named out here because #: what to do about each is the interface's rather than this sheet's: one of them closes it. @@ -5320,7 +5299,6 @@ def _fill(self) -> None: _MODEL = "model" _EFFORT = "effort" _SWARM = "swarm" -_WHERE = "where" #: Which of them are stepped along where they stand rather than opened, and which are opened. _STEPPED = (_EFFORT, _SWARM) @@ -5329,16 +5307,11 @@ def _fill(self) -> None: class Agent(Drafts[Runs]): """Everything one agent is, on one sheet, each row opened or stepped where it stands. - Which is the walk of three sheets that used to ask it, folded into the thing it was asking - about. An agent is a CLI, an account, a model at an effort and the machine its work lands - on -- and asking that as a walk meant that changing the effort of an agent already set up - was four keypresses through two sheets that had nothing to say. - - Four rows and not a dozen, because the rest of what this sheet used to ask is not the - agent's to hold. What it may do, which goals it may reach for and whether it searches the - web are the flow's, said where the flow declares the place this agent fills; the skills it - carries are its CLI's, installed and switched off where that CLI keeps them. A row - offering to set any of those would be a second answer to a question already settled. + An agent is a CLI, an account, and a model at an effort -- the word `-a` takes after the + role -- and nothing else. What it may do, what it is capable of and the skills it carries + are the flow's, declared where the flow declares the role this agent fills; where its work + lands is the environment the flow opens its session in. A row offering to set any of + those would be a second answer to a question already settled. The order the rows go in is still the order of what depends on what: the CLI settles which accounts there are to choose from and which models that CLI will name, and the account @@ -5367,24 +5340,24 @@ def __init__( runs: Runs, agents: dict[str, tuple[Model, ...]], *, - place: Place, + role: AgentRole | None = None, unavailable: frozenset[str] = frozenset(), ) -> None: """Initializes the sheet on what the agent is now. Args: - named: What to call the agent being set up, which the question at the top says. + named: The role being set up, which the question at the top says. runs: What it is now, which every row reads back. agents: The backends offered here, and what each of them says it runs. - place: What the flow declared about this one. Always one: an agent belongs to the - flow that drives it, so there is no agent here with no place to fill. + role: What the flow declared of the role, which is what rules a CLI out, or None + for an agent that is no flow's. unavailable: The optional backends that still need installing. """ super().__init__() self._named = named self._agents = dict(agents) self._unavailable = unavailable - self._place = place + self._role = role cli, _, rest = runs.spec.partition("/") model, _, effort = rest.rpartition(":") # Said outright, all of them: each is read where it is set -- what a CLI runs is @@ -5397,11 +5370,6 @@ def __init__( self._swarm: bool = effort.startswith(SWARM) self._effort: str = effort.removeprefix(SWARM) self._provider: str = runs.provider - self._anchor = runs.anchor - #: What it was handed, which is where everything this sheet does not ask comes back - #: from. Those are the flow's answers, and carrying them across is how they stay the - #: flow's rather than being reset by a sheet that never showed them. - self._given = runs #: What the chosen CLI says it runs as the chosen account, read once per pair: this #: is redrawn each time the cursor moves, and reading it is reading a file. self._catalogue: tuple[Model, ...] | None = None @@ -5425,8 +5393,7 @@ def _rows(self) -> list[tuple[str, str, str]]: Returns: One `(id, what it is set to, the line about it)` apiece, in the order they are - asked. A row nobody is being asked about is not among them: a flow that settled - where its agent works has not left that question open. + asked. The fleet row only for a model that runs a turn as one. """ rows: list[tuple[str, str, str]] = [ (_CLI, self._cli or "—", "which coding agent takes its turns"), @@ -5438,16 +5405,6 @@ def _rows(self) -> list[tuple[str, str, str]]: rows.append( (_SWARM, _YES if self._swarm else _NO, "one turn run as a fleet") ) - if pointed(self._place): - rows.append( - ( - _WHERE, - self._anchor or "this machine", - "the machine its work lands on", - ) - ) - elif image := _settled(self._place): - rows.append((_WHERE, f"in a container of {image}", "the flow settled this")) return rows def _fill(self) -> None: @@ -5546,31 +5503,14 @@ def _swarms(self) -> bool: model = self._under_model() return model is not None and model.swarms - def _tellable(self) -> bool: - """Whether the chosen CLI can be told whether its agents may search the web.""" - from hmz.coganchor.backends import named - - profile = named(self._cli) if self._cli else None - return profile is not None and profile.searches - def _made(self) -> Runs: """This agent as it now stands, which is what the sheet answers with.""" # `swarm` in front of the effort is how a fleet is asked for: one turn at one effort, # run wide. A model that does not take it is asked for at the effort alone. wide = SWARM if self._swarm and self._swarms() else "" - # On for a CLI that cannot be told, whatever the flow asked for: an agent whose - # backend has no way of being told is one that searches the web, and a config saying - # otherwise is one that backend would refuse. Only an off is turned back, though -- - # an agent nobody was asked about goes on being one nobody was asked about, that - # being a config no backend refuses and the one every CLI can serve. - searches = self._given.web_search - if searches is False and not self._tellable(): - searches = True - return self._given._replace( - spec=f"{self._cli}/{self._model}:{wide}{backends.written(self._effort)}", - anchor=self._anchor, - provider=self._provider, - web_search=searches, + return Runs( + f"{self._cli}/{self._model}:{wide}{backends.written(self._effort)}", + self._provider, ) def applied(self) -> None: @@ -5638,7 +5578,7 @@ def _took(self, event: OptionList.OptionSelected) -> None: if held == _SAVE: self.applied() return - if held in (_CLI, _ACCOUNT, _MODEL, _WHERE): + if held in (_CLI, _ACCOUNT, _MODEL): self._opens(held) @work @@ -5658,8 +5598,6 @@ async def _opens(self, held: str) -> None: await self._chose_account(showing) elif held == _MODEL: await self._chose_model(showing) - elif held == _WHERE: - await self._chose_where(showing) self._fill() async def _chose_cli(self, showing: App[None]) -> None: @@ -5668,7 +5606,7 @@ async def _chose_cli(self, showing: App[None]) -> None: Clis( self._agents, self._cli, - place=self._place, + role=self._role, unavailable=self._unavailable, ) ) @@ -5712,26 +5650,14 @@ async def _chose_model(self, showing: App[None]) -> None: self._effort = efforts[0] if efforts else "" self.changed() - async def _chose_where(self, showing: App[None]) -> None: - """Asks which machine its work lands on, where that is a question anybody is asked.""" - if not pointed(self._place): - self._said = "the flow settled where this one works" - return - where = await showing.push_screen_wait(Anchors(self._named, self._anchor)) - if where is None: - return - self._anchor, self._said = where, "" - self.changed() - class Clis(Picks): """Which coding agent takes one agent's turns, out of the ones that could. - Not always all of them: a flow that hangs a hook on a moment only some backends run said - so where it declared the place, and a CLI that does not run that moment is one choosing - would make the flow refuse to start. The same goes for whatever else that place says - filling it takes -- a turn it can talk to while it runs, a turn held to a shape -- which - is asked here of the backend the way a run of the flow asks it of the agent. + Not always all of them: a role typed as one harness -- `ClaudeCodeAgent` -- is that + harness and no other, and one declared with a capability -- `/goal`, steering, a hook + only some harnesses fire -- is one only the harnesses that have it can fill. A CLI that + cannot is one choosing would make the run refuse to start, so it is not offered. """ asked = "Select which coding agent takes its turns" @@ -5745,7 +5671,7 @@ def __init__( agents: dict[str, tuple[Model, ...]], current: str = "", *, - place: Place | None = None, + role: AgentRole | None = None, unavailable: frozenset[str] = frozenset(), ) -> None: """Initializes the choosing. @@ -5753,53 +5679,21 @@ def __init__( Args: agents: The backends offered here, and what each of them says it runs. current: The one it is now. - place: What the flow declared about this agent, which is what rules a CLI out, or - None where a CLI is being chosen for something that is not a flow's agent -- - the two ends of a fallback step, which a flow says nothing about. + role: What the flow declared of the role this agent fills, which is what rules a + CLI out, or None where a CLI is being chosen for something that is not a flow's + agent -- the two ends of a fallback step, which a flow says nothing about. unavailable: The optional backends that still need installing. """ super().__init__(current) self._agents = dict(agents) - self._place = place + self._role = role self._unavailable = unavailable - #: What the place asked of the agent that only a machine can answer, filled in as the - #: rows are built. Empty until then, and empty for the flows that asked correctly. - self._misplaced: frozenset[str] = frozenset() def rows(self) -> list[tuple[str, str, str]]: """Every CLI that could take this one's turns, and what each of them runs.""" - from hmz.runtime.flowing.checking import OF_AGENT, catalogue, misplaced - from hmz.runtime.flowing.driving import comes_to - - needs: frozenset[Moment] = ( - self._place.moments if self._place is not None else frozenset() - ) - pursuing = self._place is not None and self._place.goal - # And whatever else the flow said filling this place takes, asked the way a run of - # that flow asks it: a CLI offered here and then refused where the run is set up - # would be a question put to somebody who cannot answer it right. - serving: frozenset[str] = ( - self._place.needs.of_agent - if self._place is not None and self._place.needs is not None - else frozenset() - ) - # What the place asked of the agent but which only a machine can answer. No CLI comes - # to one of those, so every row would be ruled out and the person at the prompt would - # be told that nothing installed here will do -- which blames the installation for a - # flow that asked in the wrong half of `Needs`. Kept so that `nothing` can say so. - self._misplaced = misplaced(serving, OF_AGENT) - # Read once for the whole list rather than once per CLI: the catalogue is built off - # the live interface with `inspect` every time it is asked for, and asking it twelve - # times to answer one question is eleven walks of the same modules. - catalogued = catalogue() if serving else () listed: list[tuple[str, str, str]] = [] for backend in sorted(self._agents): - drives = _drives(backend) - if drives is None or not needs <= drives.moments: - continue - if pursuing and not drives.pursues: - continue - if serving and not serving <= comes_to(backend, catalogued=catalogued): + if _drives(backend) is None or not serves(backend, self._role): continue listed.append( ( @@ -5815,19 +5709,18 @@ def rows(self) -> list[tuple[str, str, str]]: return listed def nothing(self) -> str: - """Says so where the flow has ruled every backend here out, which is worth knowing. - - And says which of the two it was. A flow that asked of the agent for something only a - machine can answer rules out every CLI there is, and saying that nothing installed - here will do would send somebody to install a thirteenth. - """ + """Says so where the flow has ruled every backend here out, which is worth knowing.""" if self._rows: return "" - if self._misplaced: - named = ", ".join(sorted(self._misplaced)) + role = self._role + if role is not None and (role.harness is not None or role.capabilities): + asked = [ + *([str(role.harness)] if role.harness is not None else []), + *sorted(one.__name__ for one in role.capabilities), + ] return ( - f"this flow asks the agent for {named}, which is asked of where it works " - f"-- Needs(where={tuple(sorted(self._misplaced))!r}) -- so no CLI can answer" + f"{escape(role.name)} needs {escape(', '.join(asked))}, and no coding agent " + "installed here has that" ) return "no coding agent installed here can take this one's turns" @@ -7480,7 +7373,7 @@ def _about(self, ran: Ran) -> str: # Asked of the flow rather than read off the run, for the reason the menu asks it of # the flow: a flow is a directory on disk, and one marked resumable since that run is # one whose older runs can be picked up now. - return f"{held}{_DOT}can be picked up" if self._picks_up(ran.flow) else held + return f"{held}{_DOT}can be picked up" if self._carries_on(ran) else held def _fill(self) -> None: """Puts the runs up, marked where the cursor is.""" @@ -7563,12 +7456,23 @@ def _under(self) -> Ran | None: """The run the cursor is on, or None where the list has nothing in it.""" return next((one for one in self._ran if one.name == self._was), None) + def _carries_on(self, ran: Ran) -> bool: + """Whether one run can be carried on: its flow says so now, and it left a journal. + + Args: + ran: The run. + + Returns: + Whether picking it up would have anything to pick up from. + """ + return self._picks_up(ran.flow) and _hmz().epics.picks_up(ran.at) + def _picks_up(self, flow: str) -> bool: """Whether one flow says now that it can be picked up. Asked of the flow rather than of the run that recorded it: a flow is a directory on disk and may have been rewritten since, and what can happen next is what it says now. - Asked once per flow, since reading one means running its file. + Asked once per flow, since reading one means importing it. Args: flow: The flow, as the run named it. diff --git a/src/hmz/tui/tally.py b/src/hmz/tui/tally.py index 440179c6..8d080eae 100644 --- a/src/hmz/tui/tally.py +++ b/src/hmz/tui/tally.py @@ -222,6 +222,14 @@ def __init__(self, agents: Sequence[AgentBase], monitor: Monitor) -> None: self._reading: set[str] = set() self._stop = threading.Event() + def add(self, agent: AgentBase) -> None: + """Reads the logs of one more agent, which a run opens a session at a time. + + Args: + agent: The agent behind a session the run has just opened. + """ + self._agents.append(agent) + def watch(self) -> None: """Reads the logs for as long as the flow runs, on a thread of its own. @@ -247,7 +255,7 @@ def read(self) -> None: business reading, a row half written. What a run costs is worth nothing at the price of the run, so anything that goes wrong is left for the next read to find gone. """ - for agent in self._agents: + for agent in list(self._agents): profile = backends.named(agent.backend) if profile is None: continue diff --git a/tests/conftest.py b/tests/conftest.py index aa28c48b..223bf840 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -136,13 +136,14 @@ def _humanize_home(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: @pytest.fixture(autouse=True) def _nothing_running_yet() -> Iterator[None]: - """Starts each test with no flow running, on no branch, and leaves neither behind. + """Starts each test with no flow running and no flow module held, and leaves neither. - What is running is the process's own, and the branch this task is on is the context's. + What is running is the process's own, and so are the flow modules a run imported. Several tests here hold a flow open on purpose -- two agents working at once is what half the interface is about -- and its thread is still alive when the test lets go of it, so - the next test would find that flow running, say so on its own status line, and call its - own flows under it. + the next test would find that flow running and say so on its own status line; and a flow + directory a test wrote is imported under its own name, which the next test's flow of the + same name must not be answered with. """ _forgotten() yield @@ -151,12 +152,10 @@ def _nothing_running_yet() -> Iterator[None]: def _forgotten() -> None: """Leaves nothing of one test's flows for the next one to run under.""" - from hmz.runtime.flowing import driving + from hmz.runtime.flowing import engine, loading - driving._RUNNING.clear() - driving._CLAIMED.clear() - driving._WRITTEN.clear() - driving._ON.set(None) + engine._RUNS.clear() + loading.forget() @pytest.fixture(autouse=True, scope="session") diff --git a/tests/integration/agents/test_acp.py b/tests/integration/agents/test_acp.py index d00b0969..f057acfa 100644 --- a/tests/integration/agents/test_acp.py +++ b/tests/integration/agents/test_acp.py @@ -30,7 +30,8 @@ McpServer, driver, ) -from hmz.runtime.runner import flow_and_agents +from hmz.flows import HarnessKind +from hmz.runtime.runner import Runner, read_line from tests.stubs import ShellAgent, written if TYPE_CHECKING: @@ -180,15 +181,23 @@ def said_by(back, key): """ -#: A flow of one agent, for the line that names which CLI is to fill it. +#: A flow of one agent, for the line that names which CLI is to fill it: one turn of it. _FLOW = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow -@flow -def run(agent: AgentBase, task: str) -> None: - pass +class Agents(AgentCollection): + builder: Agent + + +class Envs(EnvCollection): + here: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def one(task, *, agents, envs, params, ctx): + session = await agents["builder"].spawn(env=envs["here"]) + return [agents["builder"].harness, await agents["builder"].run(task, session=session)] """ @@ -391,20 +400,23 @@ def test_a_line_naming_an_added_cli_is_driven_as_that_cli( a step away. """ flow = written(tmp_path / "flows", "one", _FLOW) - - _, agents, *_ = flow_and_agents( - ["-f", str(flow), "-a", f"{added}/m:as configured", "the task"] + line = read_line( + [ + "-f", + str(flow), + "-a", + f"builder={added}/m:as configured", + "-b", + "cost=1", + "hi", + ] ) - made = agents[0] - assert isinstance(made, AcpAgent) - assert made.backend == added - assert made.command == ("my-agent", "--acp") - held = made.new() - try: - assert held("hi") == "hi" - finally: - held.close() + (spec,) = line.agents + assert (spec.harness, spec.cli) == (HarnessKind.ACP, added) + assert Runner( + line.flow, agents=line.agents, budget=line.budget, workspace=tmp_path + ).run(line.task) == [HarnessKind.ACP, "hi"] def test_an_added_cli_is_called_what_it_runs( diff --git a/tests/integration/agents/test_cloning.py b/tests/integration/agents/test_cloning.py index ca6c462c..8c434655 100644 --- a/tests/integration/agents/test_cloning.py +++ b/tests/integration/agents/test_cloning.py @@ -19,7 +19,6 @@ import pytest -from hmz._legacy_flows import Agent, Driven from hmz.coganchor.agents import AgentConfig, Event, HumanAgent, Moment from hmz.coganchor.agents.skills import Loaded from tests.stubs import ShellAgent @@ -120,22 +119,3 @@ def test_a_clone_is_refused_a_config_its_backend_cannot_express() -> None: with pytest.raises(ValueError, match="service tier"): agent.clone(config=replace(agent.config, service_tier="fast")) - - -def test_what_a_flow_may_ask_of_an_agent_does_not_include_setting_it_up() -> None: - """The line between the two is who is entitled to say what an agent is. - - A flow declares `Agent` and is handed one, so what it can reach is what it may ask. The - settling is on `Driven`, which is how whoever hands an agent over holds it -- the runner - before the first turn, the calling of one flow by another, and the interface when somebody - watching a run says this agent is to go on as something else. - """ - asks = {name for name in dir(Agent) if not name.startswith("_")} - settles = {name for name in dir(Driven) if not name.startswith("_")} - asks - - assert settles == {"disable_goals", "loads", "reconfigure", "rename", "runs_on"} - assert "clone" in asks - # And the driver answers to both, which is what makes the split a contract rather than - # two names for one thing: a flow reaches half of it, and the run reaches all of it. - made = ShellAgent(CONFIG) - assert not [name for name in asks | settles if not hasattr(made, name)] diff --git a/tests/integration/agents/test_cursor.py b/tests/integration/agents/test_cursor.py index 8855cbe6..0999b1b9 100644 --- a/tests/integration/agents/test_cursor.py +++ b/tests/integration/agents/test_cursor.py @@ -396,20 +396,13 @@ def test_a_model_is_offered_at_the_rungs_this_account_lists_ids_for( def test_the_workspace_it_is_trusted_with_can_be_handed_back(cursor: _Calls) -> None: - """The one thing this driver overrules the bare command line about, and it is sayable. + """The one thing this driver overrules the bare command line about. - Trusted by default because a headless turn has nobody to answer the question; a flow - somebody is watching says so and gets Cursor's own behaviour back -- and asks beforehand - for the backend whose config has somewhere to say it as `settings:trust`, the name the - catalogue derives from the field rather than a second word minted beside it. + Trusted by default because a headless turn has nobody to answer the question; whoever is + watching says so and gets Cursor's own behaviour back. """ from dataclasses import replace - from hmz.runtime.flowing.checking import catalogue - - told = {one.name: one.backends for one in catalogue()} - assert told["settings:trust"] == frozenset({"cursor-agent"}) - CursorAgent(replace(cursors.CURSOR, trust=False)).new()("hello") (argv,) = cursor.argv() diff --git a/tests/integration/agents/test_permissions.py b/tests/integration/agents/test_permissions.py index 9a3d4e9f..d033b5d8 100644 --- a/tests/integration/agents/test_permissions.py +++ b/tests/integration/agents/test_permissions.py @@ -476,37 +476,6 @@ def test_every_backend_has_something_to_say_at_every_rung() -> None: assert rung in zcode._PERMITTED -#: A flow that says its one agent may look at anything and change nothing. -_READING = '''"""A flow whose agent reviews and does not write.""" - -from typing import Annotated - -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow - - -@flow -def run( - agents: tuple[Annotated[AgentBase, AgentDefaults(permission="read-only")]], task: str -) -> None: - agents[0](task) -''' - - -def test_the_flow_says_which_rung_its_agent_is_on(tmp_path: Path) -> None: - """Settled onto the agent before its first turn, over whatever it was made with.""" - from hmz.runtime.runner import Runner - - where = tmp_path / "reading.py" - where.write_text(_READING) - agent = CodexAgent(CodexAgentConfig(model="m", effort="high", permission="auto")) - - Runner(str(where), [agent]) - - assert agent.config.permission == "read-only" - assert unattended(agent.config.permission)["sandbox"] == "read-only" - - def test_an_agent_allowed_less_is_another_agent_at_the_same_model() -> None: """The config is frozen, so the rung is part of what the agent is.""" from dataclasses import replace @@ -693,26 +662,3 @@ def test_the_rung_below_each_rung_is_the_next_one_down( from hmz.coganchor.agents.codex import _tighter assert _tighter(refused) == instead - - -def test_a_rung_a_backend_was_told_not_to_carry_is_refused_as_a_declaration() -> None: - """Which is where a refusal about a rung belongs, and what it has always come back as. - - opencode is told what an agent may do in a table of its own, and an agent may be set up - not to have one written -- at which point the rung has nowhere to go. That is a config - refusing itself as it is built rather than a session refusing to open, and the place it - surfaces is the same one every other refusal about a declaration surfaces at. - """ - from hmz._legacy_flows import NotAFlow - from hmz.coganchor.agents import OpencodeAgent, OpencodeAgentConfig - from hmz.runtime.flowing.driving import Place, runs_at - - agent = OpencodeAgent( - OpencodeAgentConfig(model="p/m", effort="high", permission_table=False) - ) - place = Place( - name="reader", person=False, moments=frozenset(), permission="read-only" - ) - - with pytest.raises(NotAFlow, match="reader cannot be run as this flow declares"): - runs_at("flow.py", agent, place) diff --git a/tests/integration/agents/test_session_fork.py b/tests/integration/agents/test_session_fork.py index 908ce303..67e7fddc 100644 --- a/tests/integration/agents/test_session_fork.py +++ b/tests/integration/agents/test_session_fork.py @@ -225,7 +225,7 @@ def test_the_run_says_which_conversation_a_child_was_cut_from( ) -> None: """The backend's log shows a session that opened knowing things and never says whence.""" agent = OpencodeAgent(OPENCODE, name="builder") - epic = Epic("branching", [agent], "go", tmp_path) + epic = Epic("branching", "go", tmp_path) agent.epic = epic session = agent.new() session("hello") diff --git a/tests/integration/agents/test_zcode.py b/tests/integration/agents/test_zcode.py index 6ff60a10..cfb24014 100644 --- a/tests/integration/agents/test_zcode.py +++ b/tests/integration/agents/test_zcode.py @@ -795,27 +795,6 @@ def test_an_agent_handed_the_common_config_runs_at_what_was_always_sent( agent.stop() -def test_each_of_the_four_is_a_capability_a_flow_can_ask_for_beforehand() -> None: - """Humanize deciding something on ZCode's behalf is something a flow may ask about. - - The field is where the other answer is given; the name is what a place declares to be - refused an agent that has no such answer to give before its first turn. That name is - the field's own, under `settings:`, rather than a second word beside it: the catalogue - derives one per field, so a field renamed here is a name renamed there. - """ - from hmz.runtime.flowing.checking import catalogue - - told = {one.name: one.backends for one in catalogue()} - - for name in ( - "settings:titles", - "settings:native_search", - "settings:delivery", - "settings:protocol", - ): - assert "zcode" in told[name], name - - def test_a_failed_turn_says_what_zcode_said_about_it( server: _FakeServer, tmp_path: Path ) -> None: diff --git a/tests/integration/cli/test_cli.py b/tests/integration/cli/test_cli.py index 6bbfa785..e43e42fd 100644 --- a/tests/integration/cli/test_cli.py +++ b/tests/integration/cli/test_cli.py @@ -31,18 +31,16 @@ #: it opens onto is the half that runs on a target where no other layer is installed. COMMANDS = [ # The two leaves that say whether humanize reports its own failures and where the answer - # is kept: a command that cannot report a crash is a crash nobody hears about. And what a - # flow is, which is where the refusal a line naming no flow is answered with is written. - # Naming it must not cost the drivers: what a flow imports from `coganchor` is fetched - # when a flow names it, not when the line is read -- which is why the facts about the - # CLIs are here and nothing else of that layer is. And the front door of the runtime, - # which is the one object every way in holds: it reaches a layer only from inside the - # call that needs it, so naming it costs nothing but itself. + # is kept: a command that cannot report a crash is a crash nobody hears about. And the + # reading of the line, which names the CLIs there are in its help -- the facts about the + # CLIs, and nothing else of that layer. Not the flow API, the engine or a driver: those + # are reached once the line is known to name a flow, and `--help` names none. And the + # front door of the runtime, which is the one object every way in holds: it reaches a + # layer only from inside the call that needs it, so naming it costs nothing but itself. ( "exec", { "hmz.coganchor.backends", - "hmz._legacy_flows", "hmz.runtime.doing", "hmz.runtime.kept", "hmz.runtime.runner", diff --git a/tests/integration/cli/test_cli_output.py b/tests/integration/cli/test_cli_output.py index 1fe4705d..6f3baf9e 100644 --- a/tests/integration/cli/test_cli_output.py +++ b/tests/integration/cli/test_cli_output.py @@ -32,11 +32,11 @@ #: A flow that drives no agents and prints, which is what a layer under one does too. Under #: `--json` there is nowhere for a line like this to land but stderr. LOUD = """ -from hmz._legacy_flows import flow +from hmz.flows import AgentCollection, EnvCollection, FlowParams, flow -@flow -def run(agents: tuple[()], task: str) -> None: +@flow(agents=AgentCollection, envs=EnvCollection, params=FlowParams) +async def loud(task, *, agents, envs, params, ctx): print("a flow said this") """ @@ -323,7 +323,7 @@ def test_the_exec_line_says_who_is_reading_the_run( monkeypatch.chdir(tmp_path) flow = str(written(tmp_path, "loud", LOUD)) - assert main(["exec", "-f", flow, "--json", "go"]) == 0 + assert main(["exec", "-f", flow, "-b", "cost=1", "--json", "go"]) == 0 said = capsys.readouterr() # A flow that drives no agents says nothing, so there is nothing to write down -- and @@ -331,5 +331,5 @@ def test_the_exec_line_says_who_is_reading_the_run( assert said.out == "" assert "a flow said this" in said.err - assert main(["exec", "-f", flow, "go"]) == 0 + assert main(["exec", "-f", flow, "-b", "cost=1", "go"]) == 0 assert "a flow said this" in capsys.readouterr().out diff --git a/tests/integration/cli/test_run_command.py b/tests/integration/cli/test_run_command.py index af98d3da..702045ad 100644 --- a/tests/integration/cli/test_run_command.py +++ b/tests/integration/cli/test_run_command.py @@ -1,8 +1,13 @@ -"""The command line: a flow file, the agents it declares, and the task they are given. +"""The command line: a flow, what each of its roles is given, its params, and its budget. -Nothing here drives a real agent. A flow is handed agents and decides for itself whether to -launch anything, so a flow that only writes down what it was given exercises the whole path -from the command line to the entry point without a turn being run. + hmz exec -f FLOW -a ROLE=CLI[@PROVIDER]/MODEL:EFFORT -e ROLE=BACKEND@PROVIDER/WORKDIR + -p KEY=VALUE -b duration=...,cost=...,output_tokens=... [--resume] [--json] TASK + +Most of what is checked here drives no agent. A flow is handed views of its drivers and decides +for itself whether to open a session, so a flow that only writes down what it was handed +exercises the whole path from the command line to the flow without a turn being taken -- and +every refusal is a usage error before any agent has started. Where a turn is taken, it is taken +by the stand-in `claude` of :mod:`tests.flows.standins`. """ from __future__ import annotations @@ -17,33 +22,65 @@ import pytest -from hmz._legacy_flows import NotAFlow from hmz.cli import main -from hmz.coganchor.agents import PERMISSIONS, UNSAID, AgentConfig +from hmz.runtime.doing.running import Run +from hmz.runtime.epic import epics, read from hmz.runtime.flowing import BUILTIN_AT, ENTRY -from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent, written +from tests.flows import standins +from tests.stubs import written -#: A flow that drives nothing and writes down what it was handed, next to its own file. AGENTS -#: is filled in per test: what a flow declares there is how many agents it takes. +#: A flow that takes no turn and writes down what it was handed, beside its own file. RECORD = """ import json import os from pathlib import Path +from typing import NotRequired + +from hmz.flows import ( + Agent, + AgentCollection, + Env, + EnvCollection, + FlowParams, + LocalEnv, + Outworlder, + flow, +) + + +class Agents(AgentCollection): + builder: Agent + reviewer: NotRequired[Agent] + human: Outworlder + + +class Envs(EnvCollection): + here: LocalEnv + there: NotRequired[Env] -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +class Params(FlowParams): + rounds: int = 1 + tags: list[str] = [] -@flow -def run(agents: tuple[AGENTS], task: str) -> None: - Path(__file__).with_suffix(".json").write_text( + +@flow(agents=Agents, envs=Envs, params=Params) +async def record(task, *, agents, envs, params, ctx): + Path(__file__).with_name("seen.json").write_text( json.dumps( { - "agents": [ - [type(a).__name__, a.config.model, a.config.effort, a.id] for a in agents - ], - "held": type(agents).__name__, + "agents": { + role: [one.harness, one.provider, one.model, one.effort] + for role, one in agents.items() + if role != "human" + }, + "away": agents["human"].away, + "envs": { + role: [one.backend, one.provider, str(one.workdir)] + for role, one in envs.items() + }, + "params": params.model_dump(), + "budget": ctx.budget.model_dump(mode="json"), "task": task, "cwd": os.getcwd(), } @@ -51,759 +88,414 @@ def run(agents: tuple[AGENTS], task: str) -> None: ) """ -#: A flow that writes down which account each of its agents was configured to run as, which -#: is what an `-a` naming a provider has to reach. -ACCOUNTS = """ -import json -from pathlib import Path +#: A flow whose role asks for what only some harnesses do. +PICKY = """ +from hmz.flows import ( + AgentCollection, + ClaudeCodeAgent, + EnvCollection, + FlowParams, + Agent, + GoalCommandAgentMixin, + flow, +) -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +class Pursues(Agent, GoalCommandAgentMixin): ... -@flow -def run(agents: tuple[AgentBase, AgentBase], task: str) -> None: - Path(__file__).with_suffix(".json").write_text( - json.dumps([agent.config.provider for agent in agents]) - ) -""" -#: A flow that declares a rung for the first of its two agents and nothing for the second, -#: and writes down what each of them ended up running at. RUNG is filled in per test. -ACCESS = """ -import json -from pathlib import Path -from typing import Annotated +class Agents(AgentCollection): + claude: ClaudeCodeAgent + pursuer: Pursues -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow - -@flow -def run( - agents: tuple[Annotated[AgentBase, AgentDefaults(permission="RUNG")], AgentBase], - task: str, -) -> None: - Path(__file__).with_suffix(".json").write_text( - json.dumps([agent.config.permission for agent in agents]) - ) +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def picky(task, *, agents, envs, params, ctx): + pass """ -#: The same flow, declaring its agents as a named tuple: as many as there are places, and what -#: each of them is for. It reaches them by name to prove it was handed the type it asked for. -NAMED = """ -import json -import os -from pathlib import Path -from typing import NamedTuple +#: A resumable flow, which takes no turn either. +KEEPS = """ +from hmz.flows import AgentCollection, EnvCollection, FlowParams, flow -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - -class Agents(NamedTuple): - builder: AgentBase - reviewer: AgentBase - - -@flow -def run(agents: Agents, task: str) -> None: - Path(__file__).with_suffix(".json").write_text( - json.dumps( - { - "agents": [ - [type(one).__name__, one.id] - for one in (agents.builder, agents.reviewer) - ], - "held": type(agents).__name__, - "task": task, - "cwd": os.getcwd(), - } - ) - ) +@flow(agents=AgentCollection, envs=EnvCollection, params=FlowParams, resumable=True) +async def keeps(task, *, agents, envs, params, ctx): + ctx.state["runs"] = (ctx.state["runs"] if "runs" in ctx.state else 0) + 1 + print(f"run {ctx.state['runs']}") """ -#: A flow that declares its agents where only a type checker looks, which is nowhere the count -#: it declares can be read back from. -UNREADABLE = """ -from __future__ import annotations - -from typing import TYPE_CHECKING - -from hmz._legacy_flows import flow +#: What every line here names for `builder`, unless it names something else. +BUILDER = "builder=claude/claude-haiku-4-5:high" -if TYPE_CHECKING: - from hmz.coganchor.agents import AgentBase +#: What a line gives a run to spend, unless it is a line about budgets. +BUDGET = ["-b", "cost=1"] -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - pass -""" - -#: The flows humanize ships, each of which shows the line that would start it. A flow is a -#: directory with an `__init__.py` in it or a file of its own, and these are looked for as -#: both: a glob for one shape is a list that quietly empties the day a flow takes the other. -PREBUILT = sorted( - ( - path if path.is_file() else path / "__init__.py" - for path in ( - Path(__file__).resolve().parents[3] / "src/hmz/_legacy_flows/builtin" - ).glob("*") - if not path.name.startswith("_") - and (path.suffix == ".py" or (path / "__init__.py").is_file()) - ), - key=lambda path: path.parts, -) +def _flow(tmp_path: Path, source: str = RECORD, name: str = "record") -> str: + """Writes a flow out as a directory and answers with its path, as a line would name it.""" + return str(written(tmp_path / "flows", name, source)) -def _named(flow: Path) -> str: - """What a shipped flow is called, which is its directory where the file is an init.""" - return flow.parent.name if flow.name == "__init__.py" else flow.stem +def _seen(tmp_path: Path, name: str = "record") -> dict[str, Any]: + """What the flow written by :data:`RECORD` was handed.""" + return json.loads((tmp_path / "flows" / name / "seen.json").read_text()) -def _flow(tmp_path: Path, source: str) -> str: - """Writes a flow file and returns its path, as the command line would be given it.""" - path = tmp_path / "flow.py" - path.write_text(source) - return str(path) +def _refused(capsys: pytest.CaptureFixture[str], *argv: str) -> str: + """Runs a line that is to be refused, and answers with what it said on the way out.""" + with pytest.raises(SystemExit) as stopped: + main(["exec", *argv]) + assert stopped.value.code == 2 + return capsys.readouterr().err -def _seen(tmp_path: Path) -> dict[str, Any]: - """Reads back what the flow written by :data:`RECORD` was handed.""" - return json.loads((tmp_path / "flow.json").read_text()) +@pytest.fixture +def here(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + """A project of its own to run in.""" + at = tmp_path / "project" + at.mkdir() + monkeypatch.chdir(at) + return at -def test_it_drives_the_flow_with_the_agents_the_command_line_names( - tmp_path: Path, +def test_it_drives_the_flow_with_what_the_line_names( + tmp_path: Path, here: Path ) -> None: - """A model may hold slashes of its own, so only the backend and the effort are split off.""" - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase, AgentBase")) - main( - [ - "exec", - "-f", - flow, - "-a", - "claude/claude-opus-4-8:high", - "-a", - "kimi/kimi-code/k3:swarmmax", - "fix the build", - ] + flow = _flow(tmp_path) + + assert ( + main( + [ + "exec", + "-f", + flow, + "-a", + BUILDER, + "-p", + "rounds=3", + "-b", + "duration=1h,cost=2.5", + "the task", + ] + ) + == 0 ) + seen = _seen(tmp_path) - assert [agent[:3] for agent in seen["agents"]] == [ - ["ClaudeCodeAgent", "claude-opus-4-8", "high"], - ["KimiCodeCLIAgent", "kimi-code/k3", "swarmmax"], - ] - assert seen["task"] == "fix the build" - assert seen["held"] == "tuple" # a flow unpacks what it was promised + assert seen["agents"] == {"builder": ["claude", "", "claude-haiku-4-5", "high"]} + # Nobody is at a prompt, so whoever is outside the run is away. + assert seen["away"] is True + # The workspace is where the line was given, and a role nobody named is not there. + assert seen["envs"] == {"here": ["local", "", str(here.resolve())]} + assert seen["params"] == {"rounds": 3, "tags": []} + assert seen["budget"]["cost"] == 2.5 + assert seen["budget"]["duration"] is not None + assert seen["task"] == "the task" + assert Path(seen["cwd"]).resolve() == here.resolve() def test_one_option_may_name_several_agents_and_every_option_adds_to_them( - tmp_path: Path, + tmp_path: Path, here: Path ) -> None: - """A comma separates agents and the option repeats: the line is one list either way.""" - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase, AgentBase, AgentBase")) - main( - [ - "exec", - "-f", - flow, - "-a", - "claude/m:high,codex/m:high", - "-a", - "kimi/m:high", - "task", - ] - ) - assert [agent[0] for agent in _seen(tmp_path)["agents"]] == [ - "ClaudeCodeAgent", - "CodexAgent", - "KimiCodeCLIAgent", - ] - + flow = _flow(tmp_path) -def test_an_agent_may_be_told_which_account_to_run_as(tmp_path: Path) -> None: - """One flow, one CLI, two accounts: which is the whole reason a provider has a name.""" - flow = _flow(tmp_path, ACCOUNTS) main( [ "exec", "-f", flow, "-a", - "claude@subscription/claude-opus-5:high", - "-a", - "claude@deepseek/claude-opus-5:high", + f"{BUILDER},reviewer=codex@work/gpt-5.5:low", + *BUDGET, "task", ] ) - assert json.loads((tmp_path / "flow.json").read_text()) == [ - "subscription", - "deepseek", - ] - - -def test_an_agent_that_names_no_account_runs_as_this_machine_does( - tmp_path: Path, -) -> None: - flow = _flow(tmp_path, ACCOUNTS) - main(["exec", "-f", flow, "-a", "claude/m:high", "-a", "codex/m:high", "task"]) - assert json.loads((tmp_path / "flow.json").read_text()) == ["", ""] - - -@pytest.mark.parametrize("permission", PERMISSIONS) -def test_the_flow_says_what_each_of_its_agents_may_do( - tmp_path: Path, permission: str -) -> None: - """The place carries the rung, and a place that said nothing settles nothing. - - So the second agent is left on no rung at all, which is what it was made with and what - `-a codex/m:high` asked for: a line that names a CLI, a model and an effort has said - nothing about permissions. - """ - flow = _flow(tmp_path, ACCESS.replace("RUNG", permission)) + together = _seen(tmp_path)["agents"] main( [ "exec", "-f", flow, "-a", - "codex/m:high,claude/m:high", + BUILDER, + "--agents", + "reviewer=codex@work/gpt-5.5:low", + *BUDGET, "task", ] ) - assert json.loads((tmp_path / "flow.json").read_text()) == [permission, UNSAID] - - -def test_a_named_tuple_says_what_each_agent_is_for_as_well_as_how_many( - tmp_path: Path, -) -> None: - """A flow that named its agents is handed the type it asked for, and they answer to it.""" - from hmz.runtime.flowing import drives - - flow = _flow(tmp_path, NAMED) - assert drives(flow) == ("builder", "reviewer") - - main(["exec", "-f", flow, "-a", "claude/m:high", "-a", "codex/m:high", "task"]) - - seen = _seen(tmp_path) - assert seen["held"] == "Agents" # the named tuple, not a plain one - # And the agents took those names, so a trace groups each one's sessions under a word - # rather than under the codename an unnamed agent draws. - assert seen["agents"] == [ - ["ClaudeCodeAgent", "builder"], - ["CodexAgent", "reviewer"], - ] + assert _seen(tmp_path)["agents"] == together + # And the account is the reviewer's own: after the `@`, before the model. + assert together["reviewer"] == ["codex", "work", "gpt-5.5", "low"] -def test_an_agent_may_name_the_place_it_fills_instead_of_waiting_its_turn( - tmp_path: Path, -) -> None: - """Named, an agent fills the place the flow calls that, whatever order the line names.""" - flow = _flow(tmp_path, NAMED) +def test_an_environment_role_is_given_where_it_is(tmp_path: Path, here: Path) -> None: + elsewhere = tmp_path / "elsewhere" + elsewhere.mkdir() main( [ "exec", "-f", - flow, + _flow(tmp_path), "-a", - "reviewer=codex/m:high,builder=claude/m:high", + BUILDER, + "-e", + f"there=local@{elsewhere}", + *BUDGET, "task", ] ) - # The line named the reviewer first, and the flow still takes its builder first. - assert _seen(tmp_path)["agents"] == [ - ["ClaudeCodeAgent", "builder"], - ["CodexAgent", "reviewer"], - ] + assert _seen(tmp_path)["envs"]["there"] == ["local", "", str(elsewhere)] @pytest.mark.parametrize( - ("said", "complaint"), + ("said", "read"), [ - ( - ["-a", "builder=claude/m:high", "-a", "codex/m:high"], - "name every agent or none", - ), - ( - ["-a", "builder=claude/m:high,typo=codex/m:high"], - "drives no agent called typo", - ), - ( - ["-a", "builder=claude/m:high,builder=codex/m:high"], - "drives one agent called builder, and the line names 2", - ), - (["-a", "builder=claude/m:high"], "also drives reviewer"), - (["-a", "builder=claude/m:high,,reviewer=codex/m:high"], "bad agent ''"), + (["rounds=4"], {"rounds": 4, "tags": []}), + (['tags=["a","b"]'], {"rounds": 1, "tags": ["a", "b"]}), + (["rounds=2,tags=[]"], {"rounds": 2, "tags": []}), ], ) -def test_the_places_a_line_names_are_read_against_what_the_flow_declares( - tmp_path: Path, - capsys: pytest.CaptureFixture[str], - said: list[str], - complaint: str, -) -> None: - """Before the first turn, for the reason a miscount is: which place is which is the work.""" - flow = _flow(tmp_path, NAMED) - - with pytest.raises(SystemExit) as stopped: - main(["exec", "-f", flow, *said, "task"]) - - assert stopped.value.code == 2 - assert complaint in capsys.readouterr().err - assert not (tmp_path / "flow.json").exists() # refused before anything was driven - - -def test_a_flow_that_calls_its_agents_nothing_is_given_them_in_its_own_order( - tmp_path: Path, capsys: pytest.CaptureFixture[str] +def test_a_param_is_read_as_the_flow_declared_it( + tmp_path: Path, here: Path, said: list[str], read: dict[str, Any] ) -> None: - """A plain tuple says how many and no more, so there is no name for a line to fill.""" - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase, AgentBase")) + argv = [one for value in said for one in ("-p", value)] - with pytest.raises(SystemExit) as stopped: - main( - [ - "exec", - "-f", - flow, - "-a", - "builder=claude/m:high,reviewer=codex/m:high", - "task", - ] - ) + main(["exec", "-f", _flow(tmp_path), "-a", BUILDER, *argv, *BUDGET, "task"]) - assert stopped.value.code == 2 - assert "declares a plain tuple and calls the agents it drives nothing" in ( - capsys.readouterr().err - ) + assert _seen(tmp_path)["params"] == read @pytest.mark.parametrize( - ("said", "reported"), + ("said", "complaint"), [ - ("cli=claude,model=m,effort=high", "cli=claude"), - ("model=m", "model=m"), - ("effort=high", "effort=high"), - ("provider=work", "provider=work"), - ("service_tier=fast", "service_tier=fast"), - ("config.model_context_window=1000000", "config.model_context_window=1000000"), + (["-p", "nope=1"], "nope"), + (["-p", "rounds=many"], "rounds"), + (["-a", "human=claude/m:high"], "filled by the runtime"), + (["-e", "here=local@/tmp"], "is the workspace"), + (["-a", "nobody=claude/m:high"], "has no agent role 'nobody'"), + (["-e", "nowhere=local@/tmp"], "has no environment role 'nowhere'"), + (["-e", "there=local@/no/such/directory"], "no directory"), ], ) -def test_the_written_out_form_is_gone_and_a_line_that_writes_it_is_told_so( - tmp_path: Path, capsys: pytest.CaptureFixture[str], said: str, reported: str -) -> None: - """`=` and `,` name the places now, so the two spellings cannot both be read.""" - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase")) - - with pytest.raises(SystemExit) as stopped: - main(["exec", "-f", flow, "-a", said, "task"]) - - assert stopped.value.code == 2 - error = capsys.readouterr().err - assert f"bad agent {reported!r}" in error - assert "is gone: an agent is written CLI[@PROVIDER]/MODEL:EFFORT" in error - - -def test_a_run_is_not_put_in_a_container_from_the_line_that_starts_it( - tmp_path: Path, capsys: pytest.CaptureFixture[str] -) -> None: - """Where an agent works is the flow's to say, and `Isolated` is where it says it.""" - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase")) - - with pytest.raises(SystemExit) as stopped: - main( - [ - "exec", - "-f", - flow, - "--container", - "python:3.12", - "-a", - "claude/m:high", - "task", - ] - ) - - assert stopped.value.code == 2 - assert "unrecognized arguments: --container" in capsys.readouterr().err - - -#: A flow that says one of the agents it drives is the person at the prompt. -PEOPLED = """ -import json -import os -from pathlib import Path -from typing import NamedTuple - -from hmz.coganchor.agents import AgentBase, HumanAgent -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - assistant: AgentBase - human: HumanAgent - - -@flow -def run(agents: Agents, task: str) -> None: - agents.human.prompting = ["", "and then this"].pop - Path(__file__).with_suffix(".json").write_text( - json.dumps( - { - "agents": [[type(a).__name__, a.id] for a in agents], - "held": type(agents).__name__, - "said": [agents.human(task), agents.human(task)], - "task": task, - "cwd": os.getcwd(), - } - ) - ) -""" - - -def test_the_person_at_the_prompt_is_an_agent_nobody_is_asked_to_configure( +def test_what_the_flow_does_not_take_is_a_usage_error( tmp_path: Path, + here: Path, + capsys: pytest.CaptureFixture[str], + said: list[str], + complaint: str, ) -> None: - """A flow says it talks to them; it is handed one, and what they answer with is typed.""" - from hmz.runtime.flowing import drives - - flow = _flow(tmp_path, PEOPLED) - # Two places, one of them the person -- so one agent is asked for and one is given. - assert drives(flow) == ("assistant",) - - main(["exec", "-f", flow, "-a", "claude/m:high", "task"]) - - seen = _seen(tmp_path) - assert seen["agents"] == [["ClaudeCodeAgent", "assistant"], ["HumanAgent", "human"]] - # Said to like any other agent, and its answer is what was typed -- then "" for a - # conversation that is over, which is what ends a flow that is one. - assert seen["said"] == ["and then this", ""] - - -#: A flow whose only side is the person at the prompt: it drives no coding agent at all, so -#: there is nothing on its line to name. -ALONE = """ -import json -from pathlib import Path -from typing import NamedTuple - -from hmz.coganchor.agents import HumanAgent -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - human: HumanAgent - - -@flow -def run(agents: Agents, task: str) -> None: - agents.human.prompting = lambda: "answered" - Path(__file__).with_suffix(".json").write_text( - json.dumps({"agents": [a.id for a in agents], "said": agents.human(task)}) + error = _refused( + capsys, "-f", _flow(tmp_path), "-a", BUILDER, *said, *BUDGET, "task" ) -""" - - -def test_a_flow_whose_only_side_is_the_person_names_no_agent_at_all( - tmp_path: Path, -) -> None: - """A line that named an agent would be naming what nobody picks. - - Nobody chooses what the person runs, so a flow whose only side is them has everything it - needs the moment it is named -- and a line that named no agent is not short of anything. - """ - from hmz.runtime.flowing import drives - - flow = _flow(tmp_path, ALONE) - assert drives(flow) == () - - main(["exec", "-f", flow, "task"]) - seen = _seen(tmp_path) - assert seen == {"agents": ["human"], "said": "answered"} + assert error.startswith("hmz exec: error:") + assert complaint in error + assert not (tmp_path / "flows" / "record" / "seen.json").exists() + assert ( + epics() == [] + ) # refused before any agent started, and before a run was written -def test_a_flow_that_chooses_nobody_is_a_miscount_rather_than_a_place_to_name( - tmp_path: Path, capsys: pytest.CaptureFixture[str] +def test_a_required_role_left_out_is_a_usage_error( + tmp_path: Path, here: Path, capsys: pytest.CaptureFixture[str] ) -> None: - """It calls its agents nothing because there are none, which is what it has to be told.""" - flow = _flow(tmp_path, ALONE) + error = _refused(capsys, "-f", _flow(tmp_path), *BUDGET, "task") - with pytest.raises(SystemExit) as stopped: - main(["exec", "-f", flow, "-a", "human=claude/m:high", "task"]) + assert "needs an agent for 'builder'" in error - assert stopped.value.code == 2 - assert "the flow drives 0 agents, 1 given" in capsys.readouterr().err - -def test_a_flow_that_does_drive_agents_still_has_to_be_given_them( - tmp_path: Path, +def test_a_run_is_given_a_budget_or_is_not_started( + tmp_path: Path, here: Path, capsys: pytest.CaptureFixture[str] ) -> None: - """Which is caught against what the flow declares, as every other miscount is.""" - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase")) + error = _refused(capsys, "-f", _flow(tmp_path), "-a", BUILDER, "task") - with pytest.raises(SystemExit) as exit_code: - main(["exec", "-f", flow, "task"]) - - assert exit_code.value.code == 2 - - -#: A flow that says one of its agents has to be one a hook can say no to, which is what -#: writing the moment beside the type in the annotation means. -DEMANDING = """ -import json -from pathlib import Path -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Moment, Verdict -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - builder: Annotated[AgentBase, Moment.PERMISSION_REQUEST] - reviewer: AgentBase - - -@flow -def run(agents: Agents, task: str) -> None: - agents.builder.hooks.on(Moment.PERMISSION_REQUEST, lambda _: Verdict(refused=True)) - Path(__file__).with_suffix(".json").write_text( - json.dumps({"agents": [[a.id] for a in agents], "task": task}) - ) -""" - - -def test_a_flow_says_what_each_agent_has_to_be_able_to_do(tmp_path: Path) -> None: - """Beside the type, where the flow declares the place -- and read back before the run.""" - from hmz.coganchor.agents import Moment - from hmz.runtime.flowing import drives, wanted - - flow = _flow(tmp_path, DEMANDING) - - assert drives(flow) == ("builder", "reviewer") - assert [place.moments for place in wanted(flow)] == [ - frozenset({Moment.PERMISSION_REQUEST}), - frozenset(), - ] - - -def test_an_agent_that_cannot_do_what_its_place_asks_is_refused_before_the_run( - tmp_path: Path, -) -> None: - """Before the first turn, for the reason the count is: not hours into a loop.""" - flow = _flow(tmp_path, DEMANDING) - - with pytest.raises(SystemExit): - main( - [ - "exec", - "-f", - flow, - "-a", - "opencode/m:high", - "-a", - "opencode/m:high", - "task", - ] - ) - - assert not (tmp_path / "flow.json").exists() # nothing ran - - # And a backend that does ask before it uses a tool is taken. - main(["exec", "-f", flow, "-a", "claude/m:high", "-a", "opencode/m:high", "task"]) - assert _seen(tmp_path)["agents"] == [["builder"], ["reviewer"]] - - -def test_what_a_place_asks_for_is_said_where_it_is_refused(tmp_path: Path) -> None: - from hmz.coganchor.agents import OpencodeAgent - - flow = _flow(tmp_path, DEMANDING) - agents = [ - OpencodeAgent(AgentConfig(model="m", effort="high")), - OpencodeAgent(AgentConfig(model="m", effort="high")), - ] - - with pytest.raises(NotAFlow, match="builder has to run PermissionRequest"): - Runner(flow, agents) - - -def test_a_plain_tuple_says_how_many_agents_and_nothing_more(tmp_path: Path) -> None: - from hmz.runtime.flowing import drives - - assert drives( - _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase, AgentBase")) - ) == ( - "", - "", - ) - - -def test_two_agents_of_one_spelling_are_two_agents(tmp_path: Path) -> None: - """An actor and the reviewer reading its work are one configuration and not one agent.""" - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase, AgentBase")) - main(["exec", "-f", flow, "-a", "claude/m:high", "-a", "claude/m:high", "task"]) - ids = {agent[3] for agent in _seen(tmp_path)["agents"]} - assert len(ids) == 2 - - -def test_the_flow_runs_where_the_command_was_given( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """And not where the flow file happens to live: the work lands in this project.""" - workspace = tmp_path / "workspace" - workspace.mkdir() - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase")) - monkeypatch.chdir(workspace) - main(["exec", "-f", flow, "-a", "claude/m:high", "task"]) - assert Path(_seen(tmp_path)["cwd"]).resolve() == workspace.resolve() + assert "is given a budget" in error + assert epics() == [] @pytest.mark.parametrize( - ("source", "complaint"), + ("agents", "complaint"), [ - ("flow = None\n", "nothing in it is marked @flow()"), ( - "from hmz._legacy_flows import flow\n\n\n@flow\ndef run(agents, task):\n pass\n", - "tuple", + ["claude=codex/gpt-5.5:low", "pursuer=claude/m:high"], + "'claude' is claude, and codex was given", + ), + ( + ["claude=claude/m:high", "pursuer=opencode/opencode/big-pickle:high"], + "needs GoalCommandAgentMixin", ), - (RECORD.replace("AGENTS", "AgentBase, ..."), "fixed length"), - (RECORD.replace("AGENTS", "AgentBase, AgentBase"), "drives 2 agents, 1 given"), - (UNREADABLE, "cannot be read here"), ], ) -def test_a_file_that_is_not_the_flow_asked_for_is_a_usage_error( - tmp_path: Path, capsys: pytest.CaptureFixture[str], source: str, complaint: str +def test_an_agent_that_cannot_be_what_its_role_asks_is_refused_before_the_run( + tmp_path: Path, + here: Path, + capsys: pytest.CaptureFixture[str], + agents: list[str], + complaint: str, ) -> None: - with pytest.raises(SystemExit) as stopped: - main(["exec", "-f", _flow(tmp_path, source), "-a", "claude/m:high", "task"]) - assert stopped.value.code == 2 - assert complaint in capsys.readouterr().err - assert not (tmp_path / "flow.json").exists() # refused before anything was driven + flow = _flow(tmp_path, PICKY, "picky") + error = _refused(capsys, "-f", flow, "-a", ",".join(agents), *BUDGET, "task") -def test_a_flow_that_is_not_there_is_a_usage_error( - tmp_path: Path, capsys: pytest.CaptureFixture[str] -) -> None: - with pytest.raises(SystemExit) as stopped: - main( - ["exec", "-f", str(tmp_path / "nowhere.py"), "-a", "claude/m:high", "task"] - ) - assert stopped.value.code == 2 - assert "nowhere.py" in capsys.readouterr().err + assert complaint in error @pytest.mark.parametrize( "spec", [ - "claude/claude-opus-4-8", - "claude", - "gemini/g:high", - "/m:high", - "claude/:high", + "builder=claude/claude-opus-4-8", "builder=claude", + "builder=gemini/g:high", + "builder=/m:high", + "builder=claude/:high", + "claude/m:high", "builder=", + "builder=claude/m:high,builder=claude/m:low", ], ) -def test_an_agent_that_is_not_cli_model_and_effort_is_a_usage_error( - tmp_path: Path, capsys: pytest.CaptureFixture[str], spec: str +def test_an_agent_that_is_not_a_role_a_cli_a_model_and_an_effort_is_a_usage_error( + tmp_path: Path, here: Path, capsys: pytest.CaptureFixture[str], spec: str ) -> None: - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase")) - with pytest.raises(SystemExit) as stopped: - main(["exec", "-f", flow, "-a", spec, "task"]) - assert stopped.value.code == 2 - assert f"bad agent {spec!r}" in capsys.readouterr().err + error = _refused(capsys, "-f", _flow(tmp_path), "-a", spec, *BUDGET, "task") + assert "-a" in error + assert not (tmp_path / "flows" / "record" / "seen.json").exists() -@pytest.mark.parametrize("rung", ["auto", ""]) -def test_an_agent_at_no_rung_is_named_with_auto_and_runs( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, rung: str + +@pytest.mark.parametrize("effort", ["auto", ""]) +def test_an_agent_at_no_effort_is_named_with_auto_and_runs( + tmp_path: Path, here: Path, effort: str ) -> None: - """`auto` is the word for no rung, and a spec round-tripped without one says the same. + """`auto` is the word for the CLI's own default, and a spec without one says the same.""" + main( + [ + "exec", + "-f", + _flow(tmp_path), + "-a", + f"builder=claude/m:{effort}", + *BUDGET, + "task", + ] + ) - A model does not always have rungs -- Cursor runs `composer-2.5` and `gemini-3.1-pro` at - one setting and no other -- and before there was a word for it such a model could not be - named on this flag at all: the grammar is `MODEL:EFFORT` and there is nothing to write. - """ - workspace = tmp_path / "workspace" - workspace.mkdir() - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase")) - monkeypatch.chdir(workspace) - main(["exec", "-f", flow, "-a", f"claude/m:{rung}", "task"]) - [[_, model, effort, _]] = _seen(tmp_path)["agents"] - assert (model, effort) == ("m", "") + assert _seen(tmp_path)["agents"]["builder"] == ["claude", "", "m", ""] -@pytest.mark.parametrize( - "said", ["permission=read-only", "permission=bypass", "web_search=off"] -) -def test_a_line_that_says_what_the_flow_says_is_a_usage_error( - tmp_path: Path, - capsys: pytest.CaptureFixture[str], - said: str, +def test_a_flow_that_is_not_there_is_a_usage_error( + tmp_path: Path, here: Path, capsys: pytest.CaptureFixture[str] ) -> None: - """And says where it is said instead, which is beside the agent the flow declares.""" - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase")) + error = _refused( + capsys, "-f", str(tmp_path / "nowhere"), "-a", BUILDER, *BUDGET, "task" + ) - with pytest.raises(SystemExit) as stopped: - main(["exec", "-f", flow, "-a", said, "task"]) + assert "nowhere" in error - assert stopped.value.code == 2 - error = capsys.readouterr().err - assert f"bad agent {said!r}" in error - assert "is the flow's to say, written beside the agent" in error - assert not (tmp_path / "flow.json").exists() + +def test_a_directory_that_holds_no_flow_is_a_usage_error( + tmp_path: Path, here: Path, capsys: pytest.CaptureFixture[str] +) -> None: + error = _refused( + capsys, + "-f", + _flow(tmp_path, "HELD = 1\n", "empty"), + "-a", + BUILDER, + *BUDGET, + "task", + ) + + assert "defines no flow" in error def test_a_flow_fails_as_it_would_anywhere_when_it_is_the_flow_that_failed( - tmp_path: Path, + tmp_path: Path, here: Path ) -> None: - """A flow whose own setup cannot find a file has not been mistyped on the command line.""" - flow = _flow(tmp_path, "open('nowhere/prompt.md')\n") - with pytest.raises(FileNotFoundError): - main(["exec", "-f", flow, "-a", "claude/m:high", "task"]) + """A flow whose own import cannot find a file has not been mistyped on the command line.""" + flow = _flow(tmp_path, RECORD.replace("import json\n", "open('nowhere.md')\n", 1)) + with pytest.raises(SystemExit) as stopped: + main(["exec", "-f", flow, "-a", BUILDER, *BUDGET, "task"]) + # Refused with the reason the import gave, before any agent started. + assert stopped.value.code == 2 -def test_a_flow_for_other_agents_than_these_is_refused_before_it_is_run( - tmp_path: Path, + failing = _flow( + tmp_path, + RECORD.replace( + " Path(__file__)", + " raise FileNotFoundError(task)\n Path(__file__)", + ), + "failing", + ) + with pytest.raises(FileNotFoundError, match="task"): + main(["exec", "-f", failing, "-a", BUILDER, *BUDGET, "task"]) + (epic,) = epics() + ran = read(epic) + assert ran is not None + assert ran.how == "failed" + + +def test_resume_picks_up_the_newest_run_and_only_of_a_flow_that_can_be( + tmp_path: Path, here: Path, capsys: pytest.CaptureFixture[str] ) -> None: - """What the usage error is made of, for a flow driven from Python instead.""" - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase, AgentBase")) - with pytest.raises(NotAFlow): - Runner(flow, [ShellAgent(AgentConfig(model="m", effort="high"))]) + keeps = _flow(tmp_path, KEEPS, "keeps") + + assert "no run here to pick up" in _refused( + capsys, "-f", keeps, *BUDGET, "--resume", "task" + ) + main(["exec", "-f", keeps, *BUDGET, "task"]) + main(["exec", "-f", keeps, *BUDGET, "--resume", "task"]) + main(["exec", "-f", keeps, *BUDGET, "task"]) + + assert capsys.readouterr().out.split("\n")[:3] == ["run 1", "run 2", "run 1"] + assert "does not say it can be picked up" in _refused( + capsys, "-f", _flow(tmp_path), "-a", BUILDER, *BUDGET, "--resume", "task" + ) def test_python_m_hmz_is_the_hmz_command( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch + tmp_path: Path, here: Path, monkeypatch: pytest.MonkeyPatch ) -> None: - flow = _flow(tmp_path, RECORD.replace("AGENTS", "AgentBase")) + flow = _flow(tmp_path) monkeypatch.setattr( - sys, "argv", ["hmz", "exec", "-f", flow, "-a", "claude/m:high", "task"] + sys, "argv", ["hmz", "exec", "-f", flow, "-a", BUILDER, *BUDGET, "task"] ) + with pytest.raises(SystemExit) as stopped: runpy.run_module("hmz", run_name="__main__") + assert stopped.value.code == 0 assert _seen(tmp_path)["task"] == "task" -@pytest.mark.parametrize("flow", PREBUILT, ids=_named) def test_every_example_runs_as_the_command_line_it_shows( - flow: Path, monkeypatch: pytest.MonkeyPatch + monkeypatch: pytest.MonkeyPatch, ) -> None: - """Each one shows an `hmz exec` line, and it is one that would start that flow.""" - shown = re.search(r"^\s*hmz exec (?:.*\\\n)*.*", flow.read_text(), re.MULTILINE) - assert shown is not None, "no `hmz exec` command line to be checked against" - monkeypatch.chdir(Path(__file__).resolve().parents[3]) + """Each flow humanize ships shows an `hmz exec` line, and it is one that would start it.""" + shipped = sorted( + path / ENTRY + for path in BUILTIN_AT.iterdir() + if not path.name.startswith("_") and (path / ENTRY).is_file() + ) + assert shipped + ran: list[str] = [] - def nothing(_self: Runner, _task: str) -> None: - """Every line is checked as far as the entry point, and no further.""" + def nothing(self: Run) -> None: + """Every line is checked as far as the flow, and no further.""" + ran.append(self.flow) - monkeypatch.setattr(Runner, "run", nothing) - main(shlex.split(shown[0].replace("\\\n", " "))[1:]) + monkeypatch.setattr(Run, "run", nothing) + monkeypatch.chdir(Path(__file__).resolve().parents[3]) + for flow in shipped: + shown = re.search(r"^\s*hmz exec (?:.*\\\n)*.*", flow.read_text(), re.MULTILINE) + assert shown is not None, f"{flow}: no `hmz exec` command line to be checked" + assert main(shlex.split(shown[0].replace("\\\n", " "))[1:]) == 0 + assert len(ran) == len(shipped) def test_a_flow_of_your_own_is_found_where_flows_live( @@ -825,16 +517,14 @@ def test_a_flow_of_your_own_is_found_where_flows_live( home, project = tmp_path / "home", tmp_path / "project" for where in (home / ".humanize/flows", project / ".humanize/flows"): where.mkdir(parents=True) - mine = RECORD.replace("AGENTS", "AgentBase") - written(home / ".humanize/flows", "yours", mine) - written(project / ".humanize/flows", "theirs", mine) - written(project / ".humanize/flows", "chat", mine) # a name humanize uses + written(home / ".humanize/flows", "yours", RECORD) + written(project / ".humanize/flows", "theirs", RECORD) + written(project / ".humanize/flows", "chat", RECORD) # a name humanize uses monkeypatch.setenv("HOME", str(home)) monkeypatch.chdir(project) - listed = found() + named = [(one.whose, one.name) for one in found()] - named = [(one.whose, one.name) for one in listed] assert ("local", "local/theirs") in named assert ("user", "user/yours") in named # Both, under names of their own: one is not offered as if it were the other. @@ -845,8 +535,6 @@ def test_a_flow_of_your_own_is_found_where_flows_live( assert find("yours") == str((home / ".humanize/flows/yours" / ENTRY).resolve()) # And a flow of humanize's own said outright is not one the project can stand in for. assert find("official/chat") == str((BUILTIN_AT / "chat" / ENTRY).resolve()) - # And it takes what the list calls one, which says which place it came from and so is the - # spelling nothing can stand in for. assert find("user/yours") == str((home / ".humanize/flows/yours" / ENTRY).resolve()) assert find("local/chat") == str( (project / ".humanize/flows/chat" / ENTRY).resolve() @@ -855,9 +543,6 @@ def test_a_flow_of_your_own_is_found_where_flows_live( assert find("~/.humanize/flows/yours") == str( (home / ".humanize/flows/yours" / ENTRY).resolve() ) - assert find(".humanize/flows/theirs") == str( - (project / ".humanize/flows/theirs" / ENTRY).resolve() - ) assert find("nowhere") == "nowhere" # a path is taken as given @@ -866,57 +551,52 @@ def test_a_flow_of_your_own_runs_by_name( ) -> None: """The point of finding it: `-f theirs` starts it, with no path said anywhere.""" project = tmp_path / "project" - (project / ".humanize/flows").mkdir(parents=True) - written( - project / ".humanize/flows", "theirs", RECORD.replace("AGENTS", "AgentBase") - ) + written(project / ".humanize/flows", "theirs", RECORD) monkeypatch.setenv("HOME", str(tmp_path / "home")) monkeypatch.chdir(project) - driven: list[str] = [] - - def record(_self: Runner, task: str) -> None: - driven.append(task) - - monkeypatch.setattr(Runner, "run", record) - - assert main(["exec", "-f", "theirs", "-a", "claude/m:high", "do it"]) == 0 - assert driven == ["do it"] - -def test_the_chat_flow_is_one_session_for_as_long_as_it_is_told_things( - monkeypatch: pytest.MonkeyPatch, -) -> None: - """Talking to a coding agent, with no loop around it: the turns are a conversation.""" - from hmz._legacy_flows.builtin.chat import Chat - from hmz._legacy_flows.builtin.chat import run as chat - from hmz.coganchor.agents import HumanAgent + assert main(["exec", "-f", "theirs", "-a", BUILDER, *BUDGET, "do it"]) == 0 - agent = ShellAgent(AgentConfig(model="m", effort="high")) - said = ["echo third", "echo second"] - # The person is an agent like any other, and what they answer with is what they typed. - person = HumanAgent() - person.prompting = said.pop + seen = json.loads((project / ".humanize/flows/theirs/seen.json").read_text()) + assert seen["task"] == "do it" - chat(Chat(agent, person), "echo first") - # One session for all three, so the agent had the earlier turns in context: a second - # would have opened a second id. And the run ended when there was nothing left to be - # told, rather than looping on nothing. - assert len(agent.opened) == 1 - assert said == [] +@pytest.fixture +def claude(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + """The stand-in `claude` on PATH, with a home of its own; its log.""" + log = standins.install(tmp_path / "bin", "claude", standins.CLAUDE) + monkeypatch.setenv("PATH", standins.path_with(tmp_path / "bin")) + monkeypatch.setenv("HOME", str(tmp_path / "home")) + monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(tmp_path / "claude-home")) + return log -def test_the_chat_flow_run_from_a_command_line_does_the_one_thing_it_was_given( - monkeypatch: pytest.MonkeyPatch, +def test_chat_runs_with_no_budget_and_does_the_one_thing_it_was_given( + here: Path, claude: Path, capsys: pytest.CaptureFixture[str] ) -> None: - """Nobody is at a prompt there, so there is nothing to wait for and it returns.""" - from hmz._legacy_flows.builtin.chat import Chat - from hmz._legacy_flows.builtin.chat import run as chat - from hmz.coganchor.agents import HumanAgent - - agent = ShellAgent(AgentConfig(model="m", effort="high")) - - # Nothing is hooked up to the person, so they answer with nothing the first time. - chat(Chat(agent, HumanAgent()), "echo once") + """Nobody is at a prompt on a command line, so there is no next thing to wait for.""" + assert ( + main( + [ + "exec", + "-f", + "chat", + "-a", + "assistant=claude/claude-haiku-4-5:low", + "Reply with the single word: hello", + ] + ) + == 0 + ) - assert len(agent.opened) == 1 + # What the turn answered is on stdout, for a script reading it. + assert "hello" in capsys.readouterr().out + said = [ + json.loads(line) for line in claude.read_text().splitlines() if "said" in line + ] + assert [one["said"] for one in said] == ["Reply with the single word: hello"] + (epic,) = epics() + ran = read(epic) + assert ran is not None + assert ran.budget is not None + assert ran.budget["cost"] == "Infinity" diff --git a/tests/integration/daemon/test_holding.py b/tests/integration/daemon/test_holding.py index 18f92891..948bdbca 100644 --- a/tests/integration/daemon/test_holding.py +++ b/tests/integration/daemon/test_holding.py @@ -432,16 +432,30 @@ def test_what_flows_are_running_is_asked_of_the_runtime_rather_than_of_the_run( Which is why nothing was registered here to say so: what is running is read out of the runtime by whoever was asked about the run. """ + import time - class Drove: - flow = "rlar" + from hmz.runtime.flowing.engine import LiveCall - monkeypatch.setattr("hmz.runtime.flowing.driving.running", lambda: (Drove(),)) + top = LiveCall("rlar:rlar", "rlar", 1, time.monotonic() - 5, 1, None) + under = LiveCall("rlar:review", "review", 2, time.monotonic(), 2, top) + monkeypatch.setattr("hmz.runtime.flowing.engine.running", lambda: (top, under)) asking = holding.terminal(joining=False) asking.says(CONTROL, {"do": "status"}) - assert json.loads(asking.hears(CONTROL))["flows"] == ["rlar"] + said = json.loads(asking.hears(CONTROL)) + assert said["flows"] == ["rlar:rlar", "rlar:review"] + # The running tree whole: each call by its ref and name, how deep, how long it has been + # going, and the call that made it by its place in the list. + (first, second) = said["calls"] + assert (first["ref"], first["name"], first["depth"], first["parent"]) == ( + "rlar:rlar", + "rlar", + 1, + None, + ) + assert first["seconds"] >= 5 + assert (second["ref"], second["depth"], second["parent"]) == ("rlar:review", 2, 0) @pytest.mark.timeout(60) @@ -465,7 +479,7 @@ def test_a_runtime_that_will_not_say_is_a_run_running_nothing_rather_than_no_ans def raising() -> tuple[object, ...]: raise RuntimeError("not today") - monkeypatch.setattr("hmz.runtime.flowing.driving.running", raising) + monkeypatch.setattr("hmz.runtime.flowing.engine.running", raising) asking = holding.terminal(joining=False) asking.says(CONTROL, {"do": "status"}) @@ -473,6 +487,7 @@ def raising() -> tuple[object, ...]: said = json.loads(asking.hears(CONTROL)) assert said["ok"] assert said["flows"] == [] + assert said["calls"] == [] @pytest.mark.timeout(60) diff --git a/tests/integration/daemon/test_opening.py b/tests/integration/daemon/test_opening.py index f912863e..53899594 100644 --- a/tests/integration/daemon/test_opening.py +++ b/tests/integration/daemon/test_opening.py @@ -132,8 +132,16 @@ def __init__(self, **said: object) -> None: def run(self) -> None: return None + def said(self) -> dict[str, object]: + return {"flow": "chat"} + + held = unittest.mock.Mock(spec=daemon.Held) with unittest.mock.patch("hmz.tui.Humanize", Stands): - cli.apart(unittest.mock.Mock(spec=daemon.Held)) + cli.apart(held) + + # And what the interface says about the run it holds is what a status question is told. + (hook,), _ = held.says.call_args + assert hook() == {"flow": "chat"} assert made["session"] is not None assert set(made) == { diff --git a/tests/integration/flows/test_async_flows.py b/tests/integration/flows/test_async_flows.py deleted file mode 100644 index d05220a1..00000000 --- a/tests/integration/flows/test_async_flows.py +++ /dev/null @@ -1,385 +0,0 @@ -"""A flow written as `async def run`, which is the other way one may be written. - -A flow is a loop over turns, and a loop that has more than one turn going at a time is a loop -that has to be able to wait for several things at once. So `run` may be a coroutine, and what -starts it does not change: `Runner.run` waits for the flow either way, and the run is written -down, checked and stopped exactly as it was. What is checked here is that both kinds of flow -are read the same, run the same, and end the same -- finished, failed or stopped by hand. -""" - -from __future__ import annotations - -import json -from typing import TYPE_CHECKING - -import pytest - -from hmz._legacy_flows import NotAFlow -from hmz.cli import main -from hmz.coganchor.agents import AgentConfig, Stopped -from hmz.runtime.epic import epics, opened -from hmz.runtime.flowing import configures, drives, wanted -from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent, events - -if TYPE_CHECKING: - from pathlib import Path - -CONFIG = AgentConfig(model="m", effort="high") - -#: A flow that is a coroutine, driving two agents at once and writing down what came back. -#: Every turn of it is awaited, which is the whole difference: `agent(task)` is the turn, -#: `await agent.aturn(task)` is the same turn with the loop handed back while it takes. -ASYNC = ''' -import asyncio -import json -from pathlib import Path -from typing import NamedTuple - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The two this drives, at once rather than one after the other.""" - - actor: AgentBase - reviewer: AgentBase - - -@flow -async def run(agents: Agents, task: str) -> None: - acted, reviewed = await asyncio.gather( - agents.actor.aturn(f"echo acted-{task}"), - agents.reviewer.aturn(f"echo reviewed-{task}"), - ) - Path(__file__).with_suffix(".json").write_text( - json.dumps({"task": task, "said": [acted, reviewed]}) - ) -''' - -#: A coroutine flow that fans one agent out over many prompts at once, which is what a batch -#: is for: one session apiece, all of them going, and the answers in the order they were asked. -FANNED = """ -import json -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -async def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - said = await agent.abatch([f"echo {task}-{at}" for at in range(12)]) - Path(__file__).with_suffix(".json").write_text(json.dumps(said)) -""" - -#: A coroutine flow that can be set up, which is read off its third argument as any flow's is. -SETTABLE = ''' -import json -from pathlib import Path - -from pydantic import BaseModel, Field - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -class Config(BaseModel): - """What this flow takes.""" - - rounds: int = Field(default=3, ge=1, le=9, description="how many times round") - - -@flow -async def run(agents: tuple[AgentBase], task: str, config: Config | None = None) -> None: - setting = config or Config() - said = [await agents[0].aturn(f"echo round-{at}") for at in range(setting.rounds)] - Path(__file__).with_suffix(".json").write_text( - json.dumps({"task": task, "rounds": setting.rounds, "said": said}) - ) -''' - -#: A coroutine flow that fails partway, which is a flow that failed and not a flow to correct. -FAILING = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -async def run(agents: tuple[AgentBase], task: str) -> None: - agents[0]("echo one") - raise ValueError("the flow itself went wrong") -""" - -#: A coroutine flow stopped by hand: the agent is told to take no further turn, and the turn -#: after that raises where the flow is waiting for it. -STOPPED = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -async def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - await agent.aturn("echo one") - agent.stop() - await agent.aturn("echo two") # raises Stopped, out of the await and out of the flow -""" - -#: A coroutine flow that runs no turn at all, and writes down what the command line handed -#: it: which agents, under what names, and the task. -RECORDING = ''' -import asyncio -import json -from pathlib import Path -from typing import NamedTuple - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The two this drives, named as the flow calls them.""" - - actor: AgentBase - reviewer: AgentBase - - -@flow -async def run(agents: Agents, task: str) -> None: - await asyncio.sleep(0) # a flow that is a coroutine, and awaits like one - Path(__file__).with_suffix(".json").write_text( - json.dumps( - { - "agents": [[type(agent).__name__, agent.id] for agent in agents], - "task": task, - } - ) - ) -''' - -#: A flow whose agents are a tuple of any length, which is no answer to how many it drives -- -#: refused for a coroutine exactly as it is for a function. -UNCOUNTED = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -async def run(agents: tuple[AgentBase, ...], task: str) -> None: - pass -""" - - -def _flow(tmp_path: Path, source: str, called: str = "flow") -> Path: - """Writes a flow out and answers with its path.""" - where = tmp_path / f"{called}.py" - where.write_text(source) - return where - - -def _said(flow: Path) -> object: - """What the flow wrote down beside itself.""" - return json.loads(flow.with_suffix(".json").read_text()) - - -def test_a_flow_may_be_a_coroutine( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, ASYNC) - agents = [ShellAgent(CONFIG), ShellAgent(CONFIG)] - - Runner(flow, agents).run("the task") - - assert _said(flow) == { - "task": "the task", - "said": ["acted-the task", "reviewed-the task"], - } - - -def test_a_coroutine_flow_says_how_many_agents_it_drives_and_what_they_are_for( - tmp_path: Path, -) -> None: - flow = _flow(tmp_path, ASYNC) - - assert drives(flow) == ("actor", "reviewer") - assert [place.name for place in wanted(flow)] == ["actor", "reviewer"] - assert configures(flow) is None - # And is held to it before its first turn, as any other flow is. - with pytest.raises(NotAFlow, match="drives 2 agents, 1 given"): - Runner(flow, [ShellAgent(CONFIG)]) - - -def test_a_coroutine_flow_that_says_nothing_about_how_many_it_drives_is_refused( - tmp_path: Path, -) -> None: - with pytest.raises(NotAFlow, match="tuple of a fixed length"): - Runner(_flow(tmp_path, UNCOUNTED), [ShellAgent(CONFIG)]) - - -def test_the_agents_of_a_coroutine_flow_are_named_by_the_places_they_fill( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, ASYNC) - agents = [ShellAgent(CONFIG), ShellAgent(CONFIG, name="mine")] - - Runner(flow, agents).run("go") - - assert agents[0].id == "actor" # named where the flow said what it was for - assert agents[1].id == "mine" # a name given is a name kept - - -def test_a_coroutine_flow_drives_its_agents_at_once( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, FANNED) - - Runner(flow, [ShellAgent(CONFIG)]).run("said") - - assert _said(flow) == [f"said-{at}" for at in range(12)] - - -def test_a_coroutine_flow_is_set_up_like_any_other( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, SETTABLE) - setting = configures(flow) - assert setting is not None - assert set(setting.model_fields) == {"rounds"} - - Runner(flow, [ShellAgent(CONFIG)], {"rounds": 2}).run("go") - - assert _said(flow) == { - "task": "go", - "rounds": 2, - "said": ["round-0", "round-1"], - } - - -def test_a_coroutine_flow_left_unset_runs_on_its_own_defaults( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, SETTABLE) - - Runner(flow, [ShellAgent(CONFIG)]).run("go") - - assert _said(flow) == { - "task": "go", - "rounds": 3, - "said": ["round-0", "round-1", "round-2"], - } - - -def test_a_coroutine_flow_is_one_epic_saying_what_each_agent_opened( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A run is a run however the flow was written: the same epic, and the same sessions.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, ASYNC) - - Runner(flow, [ShellAgent(CONFIG), ShellAgent(CONFIG)]).run("the task") - - (epic,) = epics() - assert opened(epic) == { - "actor": ["acted-the task"], - "reviewer": ["reviewed-the task"], - } - assert events(epic)[-1]["how"] == "done" - - -def test_what_a_coroutine_flow_raises_comes_out_of_the_run( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - monkeypatch.chdir(tmp_path) - - with pytest.raises(ValueError, match="the flow itself went wrong"): - Runner(_flow(tmp_path, FAILING), [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - assert events(epic)[-1]["how"] == "failed" - - -def test_a_coroutine_flow_stopped_by_hand_is_written_down_as_stopped( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - monkeypatch.chdir(tmp_path) - - with pytest.raises(Stopped): - Runner(_flow(tmp_path, STOPPED), [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - assert events(epic)[-1]["how"] == "stopped" - - -async def test_a_coroutine_flow_started_from_a_loop_of_its_own_still_runs( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Started from a thread already running a loop -- an interface, or this test. - - The flow gets a loop of its own rather than this one: a flow run on the loop it was - started from would be a flow waiting for turns that are waiting for the loop the flow is - holding, which is a run that never takes its first. - """ - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, ASYNC) - - Runner(flow, [ShellAgent(CONFIG), ShellAgent(CONFIG)]).run("the task") - - assert _said(flow) == { - "task": "the task", - "said": ["acted-the task", "reviewed-the task"], - } - - -def test_a_coroutine_flow_runs_from_the_command_line_as_any_other_does( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """The whole path: `hmz exec`, the agents the line named, and a flow that is awaited. - - No turn is run: a flow decides for itself whether to launch anything, so one that only - writes down what it was handed exercises the command line without a coding agent. - """ - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, RECORDING) - - main(["exec", "-f", str(flow), "-a", "claude/m:high", "-a", "codex/m:high", "go"]) - - assert _said(flow) == { - "agents": [["ClaudeCodeAgent", "actor"], ["CodexAgent", "reviewer"]], - "task": "go", - } - - -def test_a_flow_that_is_not_a_coroutine_is_run_exactly_as_it_was( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """The other half of it: adding one kind of flow may not have moved the other.""" - monkeypatch.chdir(tmp_path) - flow = _flow( - tmp_path, - """ -import json -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - Path(__file__).with_suffix(".json").write_text( - json.dumps(agents[0](f"echo {task}")) - ) -""", - ) - - Runner(flow, [ShellAgent(CONFIG)]).run("plainly") - - assert _said(flow) == "plainly" diff --git a/tests/integration/flows/test_builtin_flows.py b/tests/integration/flows/test_builtin_flows.py index 5eee200f..06ddcb59 100644 --- a/tests/integration/flows/test_builtin_flows.py +++ b/tests/integration/flows/test_builtin_flows.py @@ -4,79 +4,151 @@ what humanize does before anything has been fetched. Everything else humanize offers is in that repository, and what those flows do is tested where they live. -Nothing here starts a coding agent. What a turn comes to is the stand-in's shell command, so a -task that exits non-zero is a turn that could not be taken -- which is the one thing this flow -has to tell apart from a conversation that is over. +Nothing here starts a coding agent: the agent is the stand-in Claude Code of +:mod:`tests.flows.standins`, and the person is an outworlder answering from a list -- or nobody, +which is what a command line is. """ from __future__ import annotations -import subprocess +import math from typing import TYPE_CHECKING import pytest -from hmz._legacy_flows.builtin import chat -from hmz.coganchor.agents import AgentConfig -from hmz.runtime.epic import STATE, epics -from hmz.runtime.flowing import resumes +from hmz.flows import AskUserHookAgentMixin, Budget, HarnessError, HarnessKind +from hmz.runtime.epic import RESUME, epics, read +from hmz.runtime.flowing import HARNESS_CAPABILITIES, builtin, resolved +from hmz.runtime.flowing.fakes import ( + FakeAgentDriver, + FakeOutworlder, + FakeSession, + run_fake, +) from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent +from tests.flows import standins if TYPE_CHECKING: from pathlib import Path -CONFIG = AgentConfig(model="m", effort="high") +#: What the agent runs, as `-a` spells it after the role. +AGENT = "claude/claude-haiku-4-5:low" + + +@pytest.fixture +def claude(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """A stand-in Claude Code on PATH, with a home of its own, run from a project of its own.""" + standins.install(tmp_path / "bin", "claude", standins.CLAUDE) + monkeypatch.setenv("PATH", standins.path_with(tmp_path / "bin")) + monkeypatch.setenv("HOME", str(tmp_path / "home")) + monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(tmp_path / "claude-home")) + (tmp_path / "project").mkdir() + monkeypatch.chdir(tmp_path / "project") + + +def test_chat_is_humanize_s_own_and_takes_whichever_agent_it_is_given() -> None: + """One agent role any harness fills, and whoever is outside the run; nothing to resume.""" + flow = resolved("chat") + declared = flow.describe() + + assert builtin(flow) + assert not declared.resumable + assert [(one.name, one.auto, one.capabilities) for one in declared.agents] == [ + ("assistant", False, frozenset()), + ("human", True, frozenset()), + ] + assert [(one.name, one.auto) for one in declared.envs] == [("workspace", True)] + + +@pytest.mark.parametrize( + ("harness", "put"), + [ + (HarnessKind.CLAUDE, ["left"]), + (HarnessKind.KIMI, ["left"]), + (HarnessKind.CODEX, ["left"]), + ], +) +async def test_a_question_the_agent_asks_is_put_to_the_person( + harness: HarnessKind, put: list[str] +) -> None: + """On a harness that asks, the hook chat hangs puts the question to the outworlder. -#: What the agent is set to do, which the stand-in runs as the shell command it is. -TASK = "echo working" + Which chat can only do because it is handed its harness's full view: it declares a plain + `Agent`, and the hook is `AskUserHookAgentMixin`'s. + """ + asked: list[str | None] = [] -#: The conversation, by the name `-f` takes. Imported above rather than written out here: a -#: flow is loaded from its file when it is run, and a module nothing imported is one coverage -#: never sees run -- so the flow the interface opens on would read as untouched while these -#: tests were driving it. -CHAT = chat.__name__.rpartition(".")[2] + async def asks(prompt: str, *, session: FakeSession, **_: object) -> str: + del prompt + asked.append(await session.ask("Which way?", ("left", "right"))) + return "done" -#: A turn that cannot be taken at all, which is what an account the backend refused looks like -#: from inside a flow. -FAILS = "exit 1" + agent = FakeAgentDriver(harness, reply=asks) + person = FakeOutworlder(["left", ""]) + await run_fake( + resolved("chat"), "go", agents={"assistant": agent}, outworlder=person + ) -@pytest.mark.timeout(60) -def test_a_conversation_is_not_a_thing_to_carry_on( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Chat is a conversation with the person, and there is no picking one of those up.""" - monkeypatch.chdir(tmp_path) + assert person.asked[0] == "Which way? (left / right)" + assert asked == put - assert not resumes(CHAT) +async def test_a_harness_that_cannot_ask_is_not_hung_a_hook_for_it() -> None: + """Full view is the harness's own: cursor-agent asks nothing, and chat hangs nothing.""" + agent = FakeAgentDriver(HarnessKind.CURSOR_AGENT, reply="done") + person = FakeOutworlder([""]) -@pytest.mark.timeout(60) -def test_a_run_of_chat_leaves_nothing_behind_to_pick_up( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """What was said is the backend's log, and nobody is at the prompt to say more.""" - monkeypatch.chdir(tmp_path) + await run_fake( + resolved("chat"), "go", agents={"assistant": agent}, outworlder=person + ) - Runner(CHAT, [ShellAgent(CONFIG)]).run(TASK) + assert AskUserHookAgentMixin not in HARNESS_CAPABILITIES[HarnessKind.CURSOR_AGENT] + assert person.asked == ["done"] - (epic,) = epics() - assert not (epic / STATE).exists() +async def test_chat_goes_on_for_as_long_as_the_person_answers() -> None: + """Each answer is the next turn of the one conversation, and nothing said ends it.""" + agent = FakeAgentDriver(reply=["one", "two", "three"]) + person = FakeOutworlder(["more", "and more", ""]) -@pytest.mark.timeout(60) -def test_a_chat_whose_opening_turn_cannot_be_taken_says_so( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: + await run_fake( + resolved("chat"), "hello", agents={"assistant": agent}, outworlder=person + ) + + assert agent.prompts == ["hello", "more", "and more"] + assert person.asked == ["one", "two", "three"] + assert len(agent.sessions) == 1 + + +def test_chat_runs_with_no_budget_of_its_own(claude: None, tmp_path: Path) -> None: + """Under `Budget(cost=inf)`, the one flow that may be; and nobody here ends it at once.""" + runner = Runner("chat", agents={"assistant": AGENT}) + + assert runner.budget == Budget(cost=math.inf) + assert runner.run("Reply with the single word: hello") is None + (epic,) = epics() + ran = read(epic) + assert ran is not None + assert ran.how == "done" + assert ran.budget == { + "duration": None, + "cost": "Infinity", + "output_tokens": None, + "graceful": True, + } + assert [one.agent for one in ran.sessions] == ["assistant"] + # What was said is the backend's log, and there is nothing to pick up. + assert not (epic / RESUME).exists() + + +def test_a_chat_whose_opening_turn_cannot_be_taken_says_so(claude: None) -> None: """The first turn is the one that fails out loud, and the reason is the loop's own shape. - Suppressed, a turn that failed answers with nothing -- and nothing is what the person + Swallowed, a turn that failed answers with nothing -- and nothing is what the person answers with where nobody is at a prompt, which this flow reads as a conversation that is - over. So a refused account and a finished conversation would be the same run: no output, - no error and a clean exit, on a flow that had done none of what it was asked. + over. So a refused turn and a finished conversation would be the same run: no output, no + error and a clean exit, on a flow that had done none of what it was asked. """ - monkeypatch.chdir(tmp_path) - - with pytest.raises(subprocess.CalledProcessError): - Runner(CHAT, [ShellAgent(CONFIG)]).run(FAILS) + with pytest.raises(HarnessError, match="no such thing"): + Runner("chat", agents={"assistant": AGENT}).run("fail: no such thing") diff --git a/tests/integration/flows/test_calling_flows.py b/tests/integration/flows/test_calling_flows.py deleted file mode 100644 index 20c11120..00000000 --- a/tests/integration/flows/test_calling_flows.py +++ /dev/null @@ -1,896 +0,0 @@ -"""One flow reaching for another by name, running it, and being seen to. - -A flow is a loop over agents, and a loop worth having is one another loop can reach for. So a -flow asks for one the way a person does -- by the name `-f` takes -- and is handed the flow's -own function to run with the agents it already has. What is running is written down as it -happens, since a flow is a Python file and nothing can ask it what it is doing. -""" - -from __future__ import annotations - -from typing import TYPE_CHECKING - -import pytest - -from hmz._legacy_flows import NotAFlow, load, running -from hmz.coganchor.agents import UNSAID, AgentConfig -from hmz.coganchor.agents.skills import Loaded -from hmz.runtime.epic import JOURNAL, epics, read, records, sessions -from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent, events, written - -if TYPE_CHECKING: - from pathlib import Path - -CONFIG = AgentConfig(model="m", effort="high") - -#: The flow being called: one agent, and it says what it was given. -INNER = '''"""The one that is called.""" - -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - Path("inner.txt").write_text(task) - agent.new()("echo " + task) -''' - -#: The one that calls it, and then does something of its own. -OUTER = '''"""The one that calls.""" - -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow -from hmz._legacy_flows import load, running - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - Path("running.txt").write_text(",".join(one.flow for one in running())) - load("inner")(agents, f"inner: {task}") - Path("outer.txt").write_text(task) -''' - - -#: One that calls the same flow twice, since two calls of one flow are two runs of it. -TWICE = '''"""The one that calls the same flow twice.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow -from hmz._legacy_flows import load - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - load("inner")(agents, "once") - load("inner")(agents, "again") -''' - -#: One that calls the one in the middle, so that the run is three flows deep. -NESTS = '''"""The one at the top.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow -from hmz._legacy_flows import load - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - load("deeper")(agents, "mid") -''' - -#: One that opens a session of its own and then calls a flow that opens one too. -DEEPER = '''"""The one in the middle.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow -from hmz._legacy_flows import load - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent.new()("echo middle") - load("inner")(agents, "deep") -''' - - -@pytest.fixture(autouse=True) -def flows(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: - """A project with flows of its own that call each other, and a home nothing wrote to.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - where = tmp_path / "project" - (where / ".humanize/flows").mkdir(parents=True) - written(where / ".humanize/flows", "inner", INNER) - written(where / ".humanize/flows", "outer", OUTER) - written(where / ".humanize/flows", "twice", TWICE) - written(where / ".humanize/flows", "deeper", DEEPER) - written(where / ".humanize/flows", "nests", NESTS) - monkeypatch.chdir(where) - return where - - -def test_a_flow_runs_the_flow_it_asked_for(flows: Path) -> None: - """Which is the whole of it: a name in, a flow to run out.""" - Runner("outer", [ShellAgent(CONFIG)]).run("do it") - - assert (flows / "inner.txt").read_text() == "inner: do it" - assert (flows / "outer.txt").read_text() == "do it" # and it carried on afterwards - - -def test_what_is_running_is_the_one_that_was_started_and_what_it_called( - flows: Path, -) -> None: - """A flow that called another must not read as the flow somebody chose.""" - Runner("outer", [ShellAgent(CONFIG)]).run("do it") - - # Written from inside the outer flow, before it called the inner one. - assert (flows / "running.txt").read_text() == "outer" - assert running() == () # and nothing is left behind when the run is over - - -def test_the_called_flow_is_running_while_it_runs(flows: Path) -> None: - """Read from inside it, which is the only moment it is true.""" - written( - flows / ".humanize/flows", - "deep", - '"""Says what is running while it runs."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n" - "from hmz._legacy_flows import running\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' Path("deep.txt").write_text(" > ".join(one.flow for one in running()))\n', - ) - written( - flows / ".humanize/flows", - "over", - '"""Calls the one that says."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n" - "from hmz._legacy_flows import load\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' load("deep")(agents, task)\n', - ) - - Runner("over", [ShellAgent(CONFIG)]).run("go") - - assert (flows / "deep.txt").read_text() == "over > deep" - - -def test_a_flow_asked_for_by_a_name_nothing_answers_to_says_so_when_it_is_asked() -> ( - None -): - """At the asking rather than an hour into a loop, which is the point of asking early.""" - with pytest.raises(NotAFlow): - load("no_such_flow_anywhere") - - -def test_a_called_flow_is_handed_the_agents_it_declares(flows: Path) -> None: - """The tuple the flow declared, named where it named them.""" - written( - flows / ".humanize/flows", - "named", - '"""Two agents, and it says what it calls them."""\n\n' - "from pathlib import Path\n" - "from typing import NamedTuple\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "class Both(NamedTuple):\n" - ' """Two of them."""\n\n' - " builder: AgentBase\n" - " reviewer: AgentBase\n\n\n" - "@flow\n" - "def run(agents: Both, task: str) -> None:\n" - ' Path("named.txt").write_text(type(agents).__name__ + " " + agents.builder.id)\n', - ) - one, two = ShellAgent(CONFIG), ShellAgent(CONFIG) - - load("named")([one, two], "go") - - said = (flows / "named.txt").read_text() - assert said.startswith("Both ") - assert said.endswith(one.id) # handed over in the order they were given - - -def test_a_called_flow_given_the_wrong_number_of_agents_says_so(flows: Path) -> None: - """Before its first turn, which is where every other miscount is caught.""" - del flows - with pytest.raises(NotAFlow, match="drives 1 agents, 2 given"): - load("inner")([ShellAgent(CONFIG), ShellAgent(CONFIG)], "go") - - -def test_a_called_flow_is_set_up_the_way_a_run_of_it_is(flows: Path) -> None: - """Read back through the flow's own model, which is what refuses one it does not take.""" - written( - flows / ".humanize/flows", - "settable", - '"""Takes a setting."""\n\n' - "from pathlib import Path\n\n" - "from pydantic import BaseModel\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "class Config(BaseModel):\n" - ' """What it takes."""\n\n' - " rounds: int = 3\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str, config: Config | None = None) -> None:\n" - ' Path("settable.txt").write_text(str((config or Config()).rounds))\n', - ) - - load("settable")([ShellAgent(CONFIG)], "go", {"rounds": 9}) - assert (flows / "settable.txt").read_text() == "9" - - load("settable")( - [ShellAgent(CONFIG)], "go" - ) # and as it comes, where none was given - assert (flows / "settable.txt").read_text() == "3" - - with pytest.raises(NotAFlow, match="takes no config"): - load("inner")([ShellAgent(CONFIG)], "go", {"rounds": 9}) - - -def test_a_call_refused_leaves_the_caller_carrying_its_own_skills(flows: Path) -> None: - """A call that never happened must not leave the caller driving somebody else's skills. - - A caller may catch the refusal -- to try another config, or to go on without that flow -- - and every session it opens after that is its own flow's. - """ - card = "---\nname: {name}\ndescription: does a thing\n---\n\nDo it.\n" - written( - flows / ".humanize/flows", - "deep", - INNER, - {"deep-notes": card.format(name="deep-notes")}, - ) - agent = ShellAgent(CONFIG) - agent.loads([Loaded("mine", flows)]) - - with pytest.raises(NotAFlow, match="takes no config"): - load("deep")([agent], "go", {"rounds": 9}) - - assert [one.name for one in agent.loaded] == ["mine"] - - -def test_a_flow_that_calls_one_written_as_a_coroutine_awaits_it(flows: Path) -> None: - """A flow answers with whatever it answers with, so an async one is awaited by its caller.""" - written( - flows / ".humanize/flows", - "slow", - '"""A flow written as a coroutine."""\n\n' - "import asyncio\n" - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n" - "from hmz._legacy_flows import running\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - " await asyncio.sleep(0)\n" - ' Path("slow.txt").write_text(",".join(one.flow for one in running()))\n', - ) - written( - flows / ".humanize/flows", - "waits", - '"""Waits for the one it called."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n" - "from hmz._legacy_flows import load\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' await load("slow")(agents, task)\n' - ' Path("after.txt").write_text("and then this")\n', - ) - - Runner("waits", [ShellAgent(CONFIG)]).run("go") - - # It really was awaited: the file the called flow writes is there, and it was running - # under the flow that called it while it wrote. - assert (flows / "slow.txt").read_text() == "waits,slow" - assert (flows / "after.txt").read_text() == "and then this" - assert running() == () - - -def test_the_run_writes_down_the_flow_it_called(flows: Path) -> None: - """An epic is what a run was, and a flow that called another is part of what it was.""" - Runner("outer", [ShellAgent(CONFIG)]).run("do it") - - (epic,) = epics() - called = [one for one in events(epic) if one["event"] in ("called", "returned")] - - assert [one["event"] for one in called] == ["called", "returned"] - assert called[0]["flow"] == "inner" - assert called[0]["task"] == "inner: do it" - - -def test_a_called_flow_is_written_down_in_a_record_of_its_own(flows: Path) -> None: - """A flow that called another is two flows, and each of them is a flow that ran.""" - Runner("outer", [ShellAgent(CONFIG)]).run("do it") - - (epic,) = epics() - ran = read(epic) - assert ran is not None - (call,) = ran.called - - assert (call.flow, call.task) == ("inner", "inner: do it") - # Named for the flow and for this call of it, and beside the record of the run itself. - assert call.record.startswith("epic.inner_") - assert call.record.endswith(".jsonl") - assert records(epic) == [epic / JOURNAL, epic / call.record] - assert call.began - assert call.ended - # And it says what it was a run of, under the record that called it. - began, *_, ended = events(epic / call.record) - assert (began["event"], began["flow"], began["task"]) == ( - "began", - "inner", - "inner: do it", - ) - assert began["under"] == JOURNAL - assert (ended["event"], ended["how"]) == ("ended", "done") - - -def test_what_a_called_flow_opened_is_written_where_it_ran(flows: Path) -> None: - """A session opened inside a called flow is that flow's, and a run has to say which.""" - Runner("outer", [ShellAgent(CONFIG)]).run("do it") - - (epic,) = epics() - ran = read(epic) - assert ran is not None - (call,) = ran.called - - # Nothing of the called flow's own in the run's record but the call itself. - assert [one["event"] for one in events(epic)] == [ - "began", - "called", - "returned", - "ended", - ] - assert [one["event"] for one in events(epic / call.record)] == [ - "began", - "opened", - "ended", - ] - # And the run is still every session it opened, each saying which flow opened it. - (one,) = sessions(epic) - assert (one.ident, one.flow) == ("inner: do it", "inner") - - -def test_a_flow_called_twice_is_written_down_twice(flows: Path) -> None: - """Two calls of one flow are two runs of it, each with sessions of its own.""" - Runner("twice", [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - ran = read(epic) - assert ran is not None - once, again = ran.called - - assert (once.flow, once.task) == ("inner", "once") - assert (again.flow, again.task) == ("inner", "again") - assert once.record != again.record - assert sorted(one.name for one in records(epic)) == sorted( - [JOURNAL, once.record, again.record] - ) - assert [one.ident for one in sessions(epic)] == ["once", "again"] - - -def test_a_flow_a_called_flow_called_is_written_down_under_it(flows: Path) -> None: - """A run is the shape it ran in: what called what, and not one flat list of it all.""" - Runner("nests", [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - ran = read(epic) - assert ran is not None - - # The run called `deeper` and nothing else: what `deeper` called is `deeper`'s to say. - assert [one.flow for one in ran.called] == ["deeper"] - (deeper,) = ran.called - (deep,) = [one for one in events(epic / deeper.record) if one["event"] == "called"] - assert deep["flow"] == "inner" - assert events(epic / str(deep["epic"]))[0]["under"] == deeper.record - # And every session of the run is still the run's, each under the flow that opened it. - assert [(one.ident, one.flow) for one in sessions(epic)] == [ - ("middle", "deeper"), - ("deep", "inner"), - ] - - -def test_a_call_that_raised_says_so_where_it_was_written(flows: Path) -> None: - """A record closes saying how what it is a record of ended, a call as much as a run.""" - written( - flows / ".humanize/flows", - "bad", - '''"""Raises.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - raise RuntimeError("no") -''', - ) - written( - flows / ".humanize/flows", - "catches", - '''"""Calls the one that raises, and carries on.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow -from hmz._legacy_flows import load - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - try: - load("bad")(agents, task) - except RuntimeError: - pass -''', - ) - - Runner("catches", [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - ran = read(epic) - assert ran is not None - (call,) = ran.called - - assert ran.how == "done" # the run went on: what failed was the flow it called - assert call.ended - last = events(epic / call.record)[-1] - assert (last["event"], last["how"]) == ("ended", "failed") - - -def test_a_flow_that_fails_is_no_longer_running(flows: Path) -> None: - """However it ends: a list of what is running that grew would say the run never stopped.""" - written( - flows / ".humanize/flows", - "bad", - '"""Raises."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' raise RuntimeError("no")\n', - ) - written( - flows / ".humanize/flows", - "tries", - '"""Calls the one that raises, and lets it through."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n" - "from hmz._legacy_flows import load\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' load("bad")(agents, task)\n', - ) - - with pytest.raises(RuntimeError, match="no"): - Runner("tries", [ShellAgent(CONFIG)]).run("go") - - assert running() == () - - -def test_a_flow_rewritten_between_calls_is_the_one_that_runs_next(flows: Path) -> None: - """Which is what makes a loop that improves its own flows a loop that then runs them. - - A flow is a directory on disk, read by running it, and `load` reads it again at every - call rather than holding the function it found the first time. So a flow rewritten while - the run is going -- by hand, or by an agent this very flow is driving -- is run as it is - now. Nothing else lets a run improve the thing it is being run by. - """ - written( - flows / ".humanize/flows", - "over", - '"""Calls the same flow twice, rewriting it in between."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n" - "from hmz._legacy_flows import load\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' calling = load("inner")\n' - " calling(agents, task)\n" - ' at = Path(".humanize/flows/inner/__init__.py")\n' - " at.write_text(at.read_text().replace('Path(\"inner.txt\")', " - "'Path(\"rewritten.txt\")'))\n" - " calling(agents, task)\n", - ) - - Runner("over", [ShellAgent(CONFIG)]).run("go") - - assert (flows / "inner.txt").read_text() == "go" # the flow as it was - assert (flows / "rewritten.txt").read_text() == "go" # and as it had become - - -def test_a_called_flow_brings_its_own_skills_and_hands_the_agents_back( - flows: Path, -) -> None: - """A skill is the flow's, so the flow that called it goes on carrying its own.""" - from hmz.coganchor.agents.skills import Loaded - - card = "---\nname: {name}\ndescription: does a thing\n---\n\nDo it.\n" - written( - flows / ".humanize/flows", - "deep", - INNER, - {"deep-notes": card.format(name="deep-notes")}, - ) - written( - flows / ".humanize/flows", - "over", - '"""Says what it is carrying, before, during and after."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n" - "from hmz._legacy_flows import load\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " (agent,) = agents\n" - ' said = [",".join(one.name for one in agent.loaded)]\n' - ' load("deep")(agents, task)\n' - ' said.append(",".join(one.name for one in agent.loaded))\n' - ' Path("carried.txt").write_text("|".join(said))\n', - {"over-notes": card.format(name="over-notes")}, - ) - agent = ShellAgent(CONFIG) - - Runner("over", [agent]).run("go") - - # Its own before the call, and its own again after it. - assert (flows / "carried.txt").read_text() == "over-notes|over-notes" - assert [one.name for one in agent.loaded] == ["over-notes"] - assert isinstance(agent.loaded[0], Loaded) - - -def test_a_called_flow_says_what_its_agents_may_do_while_it_runs(flows: Path) -> None: - """And hands them back as it found them: a call is over when it returns.""" - written( - flows / ".humanize/flows", - "strict", - '''"""The called one, whose agent may look and not touch.""" - -from pathlib import Path -from typing import Annotated - -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow - - -@flow -def run( - agents: tuple[Annotated[AgentBase, AgentDefaults(permission="read-only")]], - task: str, -) -> None: - (agent,) = agents - Path("inside.txt").write_text(agent.config.permission) -''', - ) - written( - flows / ".humanize/flows", - "over", - '''"""Says what its agent was allowed, before the call and after it.""" - -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow, load - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - said = [agent.config.permission] - load("strict")(agents, task) - said.append(agent.config.permission) - Path("allowed.txt").write_text("|".join(said)) -''', - ) - agent = ShellAgent(CONFIG) - - Runner("over", [agent]).run("go") - - assert (flows / "inside.txt").read_text() == "read-only" - # Nothing at either end: the outer flow declared no rung, so the agent it drives is on - # none, and the call it makes hands back exactly what it borrowed. - assert (flows / "allowed.txt").read_text() == "|" - assert agent.config.permission == UNSAID - - -def test_a_called_flow_that_declares_nothing_cannot_loosen_what_it_was_called_at( - flows: Path, -) -> None: - """Calling a flow must not be a way of handing yourself more than you were given. - - The caller says `auto`, the callee says nothing at all, and a place that says nothing - declares the loosest rung there is. Read as a declaration that settles something, the - callee would run at `bypass` -- so a run somebody started at `read-only` would be at - `bypass` the moment it called a flow that mentioned nothing, and the flow they did not - write is where their choice went. - """ - written( - flows / ".humanize/flows", - "quiet", - '''"""Says nothing about what its agent may do, and writes down what it got.""" - -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - Path("inner.txt").write_text(agent.config.permission) -''', - ) - written( - flows / ".humanize/flows", - "over", - '''"""Runs its agent at a rung of its own, and calls the one that says nothing.""" - -from pathlib import Path -from typing import Annotated - -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow, load - - -@flow -def run(agents: tuple[Annotated[AgentBase, AgentDefaults(permission="auto")]], task: str) -> None: - (agent,) = agents - Path("outer.txt").write_text(agent.config.permission) - load("quiet")(agents, task) -''', - ) - agent = ShellAgent(CONFIG) - - Runner("over", [agent]).run("go") - - assert (flows / "outer.txt").read_text() == "auto" - assert (flows / "inner.txt").read_text() == "auto" - - -def test_a_called_flow_tightens_what_it_was_called_at_and_hands_it_back( - flows: Path, -) -> None: - """Tightening is the point of declaring; the call ends and the caller has its own back.""" - written( - flows / ".humanize/flows", - "strict", - '''"""Its agent may look and not touch, whatever the flow above it runs at.""" - -from pathlib import Path -from typing import Annotated - -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow - - -@flow -def run( - agents: tuple[Annotated[AgentBase, AgentDefaults(permission="read-only")]], - task: str, -) -> None: - (agent,) = agents - Path("inner.txt").write_text(agent.config.permission) -''', - ) - written( - flows / ".humanize/flows", - "over", - '''"""Runs at a middle rung, and says what it is at either side of the call.""" - -from pathlib import Path -from typing import Annotated - -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow, load - - -@flow -def run(agents: tuple[Annotated[AgentBase, AgentDefaults(permission="auto")]], task: str) -> None: - (agent,) = agents - said = [agent.config.permission] - load("strict")(agents, task) - said.append(agent.config.permission) - Path("outer.txt").write_text("|".join(said)) -''', - ) - agent = ShellAgent(CONFIG) - - Runner("over", [agent]).run("go") - - assert (flows / "inner.txt").read_text() == "read-only" - assert (flows / "outer.txt").read_text() == "auto|auto" - # And the run's own flow keeps what it declared: a run is not a call, and nothing above - # it is waiting to have the agent handed back. - assert agent.config.permission == "auto" - - -def test_a_call_refused_leaves_the_caller_driving_the_agents_it_had( - flows: Path, -) -> None: - """A call that never happened has changed nothing about what the caller is driving.""" - written( - flows / ".humanize/flows", - "quiet", - '''"""Two agents, and neither of them reads the internet.""" - -from typing import Annotated - -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow - -Quiet = Annotated[AgentBase, AgentDefaults(web_search=False)] - - -@flow -def run(agents: tuple[Quiet, Quiet], task: str) -> None: - for agent in agents: - agent(task) -''', - ) - written( - flows / ".humanize/flows", - "over", - '''"""Calls it with one agent that cannot be told, and carries on.""" - -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import NotAFlow, flow, load - - -@flow -def run(agents: tuple[AgentBase, AgentBase], task: str) -> None: - try: - load("quiet")(agents, task) - except NotAFlow as refused: - Path("refused.txt").write_text(str(refused)) - Path("after.txt").write_text( - ",".join(str(agent.config.web_search) for agent in agents) - ) -''', - ) - from hmz.coganchor.agents import ( - ClaudeCodeAgent, - ClaudeCodeAgentConfig, - PiAgent, - PiAgentConfig, - ) - - # One that can be told and one that cannot, in that order: the first is settled, and - # put back again when the second is refused. - told = ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high")) - cannot = PiAgent(PiAgentConfig(model="m", effort="high")) - - Runner("over", [told, cannot]).run("go") - - assert ( - "no way of being told not to search the web" - in (flows / "refused.txt").read_text() - ) - # Nobody was ever asked about either of them, and a call that never happened has not - # made either one an answer. - assert (flows / "after.txt").read_text() == "None,None" - assert told.config.web_search is None - - -def test_a_called_flow_can_explicitly_inherit_its_callers_skills(flows: Path) -> None: - """A wrapper can add a skill without copying the called flow's own bundle.""" - card = "---\nname: {name}\ndescription: does a thing\n---\n\n{says}\n" - inner = '''"""Records everything it carries.""" - -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - names = ",".join(one.name for one in agent.loaded) - shared = next(one for one in agent.loaded if one.name == "shared") - Path("inherited.txt").write_text(names + "|" + (shared.at / "SKILL.md").read_text()) -''' - written( - flows / ".humanize/flows", - "deep", - inner, - { - "deep-notes": card.format(name="deep-notes", says="Deep."), - "shared": card.format(name="shared", says="Child wins."), - }, - ) - written( - flows / ".humanize/flows", - "over", - '''"""Passes its skills into a called flow.""" - -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow -from hmz._legacy_flows import load - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - load("deep", inherit_skills=True)(agents, task) - Path("restored.txt").write_text(",".join(one.name for one in agents[0].loaded)) -''', - { - "over-notes": card.format(name="over-notes", says="Over."), - "shared": card.format(name="shared", says="Parent loses."), - }, - ) - agent = ShellAgent(CONFIG) - - Runner("over", [agent]).run("go") - - inherited = (flows / "inherited.txt").read_text() - assert inherited.startswith("deep-notes,shared,over-notes|") - assert "Child wins." in inherited - assert "Parent loses." not in inherited - assert (flows / "restored.txt").read_text() == "over-notes,shared" - assert [one.name for one in agent.loaded] == ["over-notes", "shared"] - - -def test_inherited_skills_are_restored_when_the_called_flow_raises(flows: Path) -> None: - """The template's cleanup applies to unsuccessful calls as well as returns.""" - card = "---\nname: {name}\ndescription: does a thing\n---\n\nDo it.\n" - written( - flows / ".humanize/flows", - "bad", - '''"""Fails after receiving inherited skills.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - assert [one.name for one in agents[0].loaded] == ["child", "parent"] - raise RuntimeError("no") -''', - {"child": card.format(name="child")}, - ) - written( - flows / ".humanize/flows", - "over", - '''"""Catches a failed inherited call and records its restored skills.""" - -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow -from hmz._legacy_flows import load - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - try: - load("bad", inherit_skills=True)(agents, task) - except RuntimeError: - pass - Path("restored-after-error.txt").write_text(",".join(one.name for one in agents[0].loaded)) -''', - {"parent": card.format(name="parent")}, - ) - - Runner("over", [ShellAgent(CONFIG)]).run("go") - - assert (flows / "restored-after-error.txt").read_text() == "parent" diff --git a/tests/integration/flows/test_flow_budget.py b/tests/integration/flows/test_flow_budget.py deleted file mode 100644 index 2d315fa1..00000000 --- a/tests/integration/flows/test_flow_budget.py +++ /dev/null @@ -1,238 +0,0 @@ -"""What a run may spend: what a flow declares, what a file says, and which of them wins. - -A flow never holds itself to any of this. What it may say is a default, and what settles the -run is one ranking in one place -- whoever started it, else the flow, else nothing at all -- -so that a flow started from a command line, from the menu and from another flow are all held -to the same thing. What is checked here is that ranking, that the reserved `budget:` key of a -settings file never reaches the flow's own model, and that a bare number is refused with all -three meanings it could have had rather than quietly taken for one of them. -""" - -from __future__ import annotations - -from typing import TYPE_CHECKING - -import pytest - -from hmz.coganchor.agents import AgentConfig, Allowance -from hmz.runtime.runner import Runner, flow_and_agents, set_up_from -from tests.stubs import ShellAgent, written - -if TYPE_CHECKING: - from pathlib import Path - -CONFIG = AgentConfig(model="m", effort="high") - -#: A flow with no opinion about what a run of it is worth, which is every flow written before -#: there was such a thing. -QUIET = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - pass -""" - -#: One that says what a run of it is worth by default. -SAYS = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import Allowance, flow - - -@flow(budget=Allowance(hours=6, tokens=10.0, dollars=50)) -def run(agents: tuple[AgentBase], task: str) -> None: - pass -""" - -#: And one that says it is meant to run under nothing at all, which `chat` is. -LOOSE = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import Allowance, flow - - -@flow(budget=Allowance()) -def run(agents: tuple[AgentBase], task: str) -> None: - pass -""" - - -def _runner(at: Path, model: str = "", **said: object) -> Runner: - """One loaded flow, with the agent it declares. - - Args: - at: The flow. - model: What that agent runs, or "" for `m`, which is on nobody's price list. - said: What whoever started the run said about it. - - Returns: - The runner. - """ - config = AgentConfig(model=model, effort="high") if model else CONFIG - return Runner(at, [ShellAgent(config)], **said) # pyright: ignore[reportArgumentType] - - -def test_a_flow_that_says_nothing_runs_under_nothing(tmp_path: Path) -> None: - """A shipped default would cut off every flow written before this on its first run.""" - at = written(tmp_path, "quiet", QUIET) - - runner = _runner(at) - - assert runner.budget == Allowance() - assert runner.unwatched # and so is the thing the menu asks about - - -def test_what_a_flow_declares_is_what_a_run_of_it_is_held_to(tmp_path: Path) -> None: - """Which is the default, said in the flow's own file where a reviewer reads it.""" - at = written(tmp_path, "says", SAYS) - - runner = _runner(at) - - assert runner.budget == Allowance(hours=6, tokens=10.0, dollars=50) - assert not runner.unwatched - - -def test_what_the_run_was_given_wins_over_what_the_flow_says(tmp_path: Path) -> None: - """The flow said a default; whoever started the run said what this run is worth.""" - at = written(tmp_path, "says", SAYS) - - runner = _runner(at, budget={"hours": 1}) - - assert runner.budget == Allowance(hours=1) - - -def test_a_flow_that_declares_nothing_at_all_is_not_asked_about_it( - tmp_path: Path, -) -> None: - """`Allowance()` written out is a flow saying an unbounded run is what it is for. - - Which is the whole of the exemption: `chat` is a conversation that ends when the person - stops typing, and a confirmation asked of every one of those is a confirmation nobody - reads. It is one line in one file rather than a name in a table in three places. - """ - at = written(tmp_path, "loose", LOOSE) - - runner = _runner(at) - - assert runner.budget == Allowance() - assert not runner.unwatched - - -def test_the_budget_in_a_settings_file_never_reaches_the_flows_own_model( - tmp_path: Path, -) -> None: - """The flow's model refuses a field it never declared, and this is not the flow's.""" - said = tmp_path / "setup.yaml" - said.write_text("budget:\n tokens: 25\nrounds: 12\n", encoding="utf-8") - - held, budget = set_up_from(said) - - assert held == {"rounds": 12} - assert budget == Allowance(tokens=25) - - -def test_a_budget_written_as_one_number_is_refused_with_all_three_meanings( - tmp_path: Path, -) -> None: - """`budget: 25` is what every flowverse loop's settings file used to say. - - Read as any one of the three it would be a run held to something nobody asked for -- a - quarter of a day, twenty-five million tokens or twenty-five dollars are not each other -- - so it is refused and all three are named. - """ - said = tmp_path / "setup.yaml" - said.write_text("budget: 25\n", encoding="utf-8") - - with pytest.raises(ValueError, match="rather than one number") as refused: - set_up_from(said) - - assert "tokens: 25" in str(refused.value) - assert "hours: 25" in str(refused.value) - assert "dollars: 25" in str(refused.value) - - -@pytest.mark.parametrize( - ("held", "because"), - [ - ("budget:\n weeks: 2\n", "takes hours, tokens, dollars"), - ("budget:\n hours: -1\n", "less than nothing"), - ("budget:\n hours: yes\n", "is a number"), - ], -) -def test_a_budget_that_cannot_be_read_names_the_file( - tmp_path: Path, held: str, because: str -) -> None: - """Said where it was written, rather than as a run that stops when nobody expects it.""" - said = tmp_path / "setup.yaml" - said.write_text(held, encoding="utf-8") - - with pytest.raises(ValueError, match=because): - set_up_from(said) - - -def test_the_exec_line_carries_what_the_file_said_the_run_may_spend( - tmp_path: Path, -) -> None: - """`-c` is one file and it now says two things: the flow's setup and the run's own.""" - (tmp_path / "b.yaml").write_text("budget:\n hours: 0.5\n", encoding="utf-8") - - _, _, _, held, budget, _ = flow_and_agents( - ["-f", "flow", "-c", str(tmp_path / "b.yaml"), "-a", "claude/m:high", "go"] - ) - - # Nothing at all for the flow: a file saying only what the run may spend has not set the - # flow up, and a flow that takes no config must not be handed an empty one. - assert held is None - assert budget == Allowance(hours=0.5) - - -def test_a_run_nothing_in_it_can_price_is_a_run_with_no_cap_on_it( - tmp_path: Path, -) -> None: - """A fifty-dollar limit on a model nobody lists is a run with no limit on it at all. - - And it reads exactly like a limit that has not been reached yet -- so it is said out loud - before the first turn, and counted as the no cap it is rather than as the cap it was - written down as. A benchmark ran eight cells of an unpriced model under a dollar apiece - and was told so in a log line, which is nowhere anybody was being asked anything. - """ - at = written(tmp_path, "quiet", QUIET) - - runner = _runner(at, budget={"dollars": 50}) - - assert "dollars" in runner.unreadable() - assert runner.unwatched # bounded on paper, and nothing here can read the paper - - -def test_a_cap_the_run_can_read_is_a_run_something_will_stop( - tmp_path: Path, priced: str -) -> None: - """The control: the same allowance on a model somebody lists is a cap that bites.""" - at = written(tmp_path, "quiet", QUIET) - - runner = _runner(at, priced, budget={"dollars": 50}) - - assert runner.unreadable() == "" - assert not runner.unwatched - - -def test_a_clock_beside_an_unreadable_cap_still_stops_the_run(tmp_path: Path) -> None: - """Which is exactly the case that benchmark hit, and the reason nothing broke there. - - The money could not be read and the hours could, so the cells stopped on the clock. A run - with one readable cap on it is a run something will stop, and must not be asked about. - """ - at = written(tmp_path, "quiet", QUIET) - - runner = _runner(at, budget={"hours": 0.2, "dollars": 1}) - - assert "dollars" in runner.unreadable() - assert not runner.unwatched - - -def test_a_cap_that_can_be_read_is_not_complained_about(tmp_path: Path) -> None: - """Blindness is about a cap that will not bite, and the clock always can be read.""" - at = written(tmp_path, "quiet", QUIET) - - assert _runner(at, budget={"hours": 6}).unreadable() == "" diff --git a/tests/integration/flows/test_flow_config.py b/tests/integration/flows/test_flow_config.py deleted file mode 100644 index edaf7564..00000000 --- a/tests/integration/flows/test_flow_config.py +++ /dev/null @@ -1,232 +0,0 @@ -"""A flow that says what it can be set up with, and what is done with what it says. - -The third argument of `run` is the whole of it: annotated with a pydantic model, the flow is -one that can be configured, and the model is what asks. A flow without one is every flow -written before there was such a thing, and is called with two arguments as it always was. -""" - -from __future__ import annotations - -import json -from typing import TYPE_CHECKING - -import pytest -from pydantic import BaseModel, Field - -from hmz._legacy_flows import NotAFlow -from hmz.coganchor.agents import AgentConfig -from hmz.runtime.flowing import configures -from hmz.runtime.runner import Runner, flow_and_agents, set_up_from -from tests.stubs import ShellAgent - -if TYPE_CHECKING: - from pathlib import Path - -CONFIG = AgentConfig(model="m", effort="high") - -#: A flow that can be set up, and writes down what it was set up with. -SETTABLE = ''' -import json -from pathlib import Path -from typing import Literal - -from pydantic import BaseModel, Field - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -class Config(BaseModel): - """What this flow takes.""" - - loud: bool = Field(default=False, description="say it twice") - rounds: int = Field(default=3, ge=1, le=9, description="how many times round") - mode: Literal["fast", "slow"] = Field(default="fast", description="which way") - - -@flow -def run(agents: tuple[AgentBase], task: str, config: Config | None = None) -> None: - Path(__file__).with_suffix(".json").write_text( - json.dumps({"task": task, "config": None if config is None else config.model_dump()}) - ) -''' - -#: The same flow, taking nothing at all, which is what a flow used to be. -PLAIN = """ -import json -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - Path(__file__).with_suffix(".json").write_text(json.dumps({"task": task})) -""" - -#: A flow whose third argument is not a model, which is a flow that takes no setting up. -NOT_A_MODEL = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str, config: str = "") -> None: - pass -""" - - -def _flow(tmp_path: Path, source: str) -> Path: - """Writes a flow out and answers with its path.""" - where = tmp_path / "flow.py" - where.write_text(source) - return where - - -def test_a_flow_says_what_it_can_be_set_up_with(tmp_path: Path) -> None: - """The model is read off the annotation, without the flow being run.""" - model = configures(_flow(tmp_path, SETTABLE)) - - assert model is not None - assert set(model.model_fields) == {"loud", "rounds", "mode"} - assert model.model_fields["rounds"].description == "how many times round" - - -@pytest.mark.parametrize("source", [PLAIN, NOT_A_MODEL]) -def test_a_flow_that_takes_no_setting_up_says_so(tmp_path: Path, source: str) -> None: - """Two arguments, or a third that is not a model, is a flow with nothing to ask about.""" - assert configures(_flow(tmp_path, source)) is None - - -def test_what_it_was_set_up_with_reaches_the_entry_point(tmp_path: Path) -> None: - """The instance the runner was given is the one the flow is called with.""" - where = _flow(tmp_path, SETTABLE) - model = configures(where) - assert model is not None - - Runner(where, [ShellAgent(CONFIG)], model(loud=True, rounds=7)).run("go") - - said = json.loads(where.with_suffix(".json").read_text()) - assert said["config"] == {"loud": True, "rounds": 7, "mode": "fast"} - - -def test_a_flow_left_alone_is_called_with_none(tmp_path: Path) -> None: - """Which is what the flow's own default means, and is the run nobody set up.""" - where = _flow(tmp_path, SETTABLE) - - Runner(where, [ShellAgent(CONFIG)]).run("go") - - assert json.loads(where.with_suffix(".json").read_text())["config"] is None - - -def test_a_flow_that_takes_nothing_is_called_as_it_always_was(tmp_path: Path) -> None: - """No third argument is passed, so every flow written before this still runs.""" - where = _flow(tmp_path, PLAIN) - - Runner(where, [ShellAgent(CONFIG)]).run("go") - - assert json.loads(where.with_suffix(".json").read_text()) == {"task": "go"} - - -def test_being_set_up_with_something_else_is_refused_before_anything_runs( - tmp_path: Path, -) -> None: - """A config of the wrong model is a caller to correct, not a flow to start.""" - - class Other(BaseModel): - wrong: bool = Field(default=True, description="not this flow's") - - where = _flow(tmp_path, SETTABLE) - - with pytest.raises(NotAFlow, match="the flow takes a Config"): - Runner(where, [ShellAgent(CONFIG)], Other()) - - -def test_the_fields_alone_are_enough_to_set_a_flow_up(tmp_path: Path) -> None: - """Which is what a YAML file of them is, and what `hmz exec -c` hands over.""" - where = _flow(tmp_path, SETTABLE) - - Runner(where, [ShellAgent(CONFIG)], {"rounds": 7, "mode": "slow"}).run("go") - - said = json.loads(where.with_suffix(".json").read_text()) - assert said["config"] == {"loud": False, "rounds": 7, "mode": "slow"} - - -def test_fields_the_flow_will_not_take_are_refused_before_it_runs( - tmp_path: Path, -) -> None: - """The flow's own model is what refuses them, at the moment the flow is about to run.""" - where = _flow(tmp_path, SETTABLE) - - with pytest.raises(NotAFlow, match="validation error"): - Runner(where, [ShellAgent(CONFIG)], {"rounds": 99}) - - -def test_a_config_is_read_off_a_yaml_file_as_it_is_written(tmp_path: Path) -> None: - """One field per line, under the names the flow declared, and nothing else in it.""" - said = tmp_path / "setup.yaml" - said.write_text("rounds: 7\nmode: slow\n") - - assert set_up_from(said) == ({"rounds": 7, "mode": "slow"}, None) - # An empty file is a flow left as it comes, which is what writing nothing means -- and - # `None` is what that is said with, since a flow taking no config takes no empty one. - (tmp_path / "empty.yaml").write_text("") - assert set_up_from(tmp_path / "empty.yaml") == (None, None) - - -@pytest.mark.parametrize( - ("held", "because"), - [("- one\n- two\n", "not a list"), ("just a string\n", "not a str")], -) -def test_a_config_file_that_is_not_a_mapping_is_said_to_be( - tmp_path: Path, held: str, because: str -) -> None: - """A flow is set up field by field, so a file holding anything else is one to correct.""" - said = tmp_path / "setup.yaml" - said.write_text(held) - - with pytest.raises(ValueError, match=because): - set_up_from(said) - - -def test_a_config_file_that_cannot_be_read_is_said_to_be(tmp_path: Path) -> None: - """Rather than a traceback out of the command line that named it.""" - with pytest.raises(ValueError, match="cannot read"): - set_up_from(tmp_path / "nowhere.yaml") - - -def test_the_exec_line_reads_the_config_it_names(tmp_path: Path) -> None: - """`hmz exec -c` is the whole of it: the file, unchecked, for the flow to check.""" - (tmp_path / "setup.yaml").write_text("rounds: 7\n") - - _, agents, task, held, budget, _ = flow_and_agents( - ["-f", "flow", "-c", str(tmp_path / "setup.yaml"), "-a", "claude/m:high", "go"] - ) - - assert held == {"rounds": 7} - assert budget is None # a file that says nothing about one leaves the flow's own - assert len(agents) == 1 - assert task == "go" - - -def test_the_exec_line_without_a_config_says_nothing_about_one(tmp_path: Path) -> None: - """Which is every line written before there was such a thing, and is a flow as it comes.""" - _, _, _, held, budget, _ = flow_and_agents( - ["-f", "flow", "-a", "claude/m:high", "go"] - ) - - assert held is None - assert budget is None - - -def test_a_flow_that_takes_nothing_refuses_being_set_up(tmp_path: Path) -> None: - """Handing a config to a flow that declared none is the same mistake the other way.""" - - class Other(BaseModel): - wrong: bool = True - - where = _flow(tmp_path, PLAIN) - - with pytest.raises(NotAFlow, match="the flow takes no config"): - Runner(where, [ShellAgent(CONFIG)], Other()) diff --git a/tests/integration/flows/test_flow_naming.py b/tests/integration/flows/test_flow_naming.py index e842f735..38cca63d 100644 --- a/tests/integration/flows/test_flow_naming.py +++ b/tests/integration/flows/test_flow_naming.py @@ -1,489 +1,192 @@ -"""What makes a function a flow, what it is called, and how one of them is asked for. +"""What a module of flows offers, what each flow in it is called, and how one is asked for. -A flow is a function marked with `@flow`, and nothing else is one -- not a function called -`run`, which is a name a file is free to use for anything. `@flow` is the flow its file holds -under the file's own name; `@flow(name=...)` is one of several, called `:`, so -three phases of one thing are one thing to write and three to run, each asking only for the -agents it drives and only for the settings it takes. +A flow is a function marked with `@flow`, and a flow directory is a module of them. The one a +bare name means is the one named after the directory, else the only one it does not hide; it is +listed under the directory's own name, and every other flow the module shows is listed as +`:` -- which is also how it is asked for. A hidden flow is listed nowhere, and still +answers to its name. """ from __future__ import annotations -import re -from pathlib import Path +from typing import TYPE_CHECKING import pytest -from hmz._legacy_flows import NotAFlow, flow -from hmz.coganchor.agents import AgentConfig -from hmz.coganchor.agents.codenames import SAID -from hmz.runtime.flowing import ( - ENTRY, - about, - configures, - drives, - find, - found, - held, - inside, - wanted, -) -from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent, written - -CONFIG = AgentConfig(model="m", effort="high") - -#: A flow that is two flows beside each other, neither of them the directory's own. -THREE = '''"""Three phases of one thing, which are three things to run.""" - -from typing import NamedTuple - -from pydantic import BaseModel - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import FlowNotFound +from hmz.runtime.flowing import ENTRY, LOCAL, about, find, found, resolved +from tests.stubs import written +if TYPE_CHECKING: + from pathlib import Path -class Drafting(NamedTuple): - """The one that writes.""" +#: What every flow here declares, which is nothing: what is checked is what each is called. +_HEAD = """ +from hmz.flows import AgentCollection, EnvCollection, FlowParams, flow - drafter: AgentBase +class Nothing(AgentCollection): + pass -class Building(NamedTuple): - """The one that builds, and the one that reads it.""" - builder: AgentBase - reviewer: AgentBase - - -class Wide(BaseModel): - """What the first phase takes.""" - - n: int = 6 +class Nowhere(EnvCollection): + pass +""" +#: A module named `three` whose three phases are three flows, one of them named for it. +THREE = ( + '"""Three phases of one thing, which are three things to run."""\n' + + _HEAD + + ''' -@flow(name="gen-idea") -def first_pass(agents: Drafting, task: str, config: Wide | None = None) -> None: +@flow(agents=Nothing, envs=Nowhere, params=FlowParams, name="gen-idea") +async def first_pass(task, *, agents, envs, params, ctx): """Opens a loose idea into a draft.""" - agents.drafter.new()(f"{task} {(config or Wide()).n}") + return "idea" -@flow(name="build", about="builds it, under review") -def start_it(agents: Building, task: str) -> None: +@flow(agents=Nothing, envs=Nowhere, params=FlowParams, description="builds it") +async def three(task, *, agents, envs, params, ctx): """A docstring the decorator was told to say something else instead of.""" - agents.builder.new()(task) - - -def run(agents: Drafting, task: str) -> None: - """Called run, marked with nothing, and so not a flow at all.""" -''' - -#: A file that is one flow, under a function name that says nothing about it. -ONE = '''"""Just the one, and it says what it does here.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def whatever_it_is_called(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent.new()(task) -''' - -#: And one that is both: the file's own flow, and another beside it. -BOTH = '''"""One under its own name, and one beside it.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow + return "three" -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - """What the file itself is.""" - (agent,) = agents - agent.new()(task) +@flow(agents=Nothing, envs=Nowhere, params=FlowParams, hidden=True) +async def engine(task, *, agents, envs, params, ctx): + """What the others call, and nobody picks.""" + return "engine" -@flow(name="twice") -def twice(agents: tuple[AgentBase], task: str) -> None: - """The other one.""" - (agent,) = agents - agent.new()(task) - agent.new()(task) +async def run(task): + """Called run, marked with nothing, and so not a flow at all.""" ''' +) -#: A public composition and the internal engine it calls by name. -AUXILIARY = '''"""One flow to choose and one implementation detail.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent.new()(task) - +#: A module that is one flow, under a function name that says nothing about it. +ONE = ( + '"""Just the one, and it says what it does here."""\n' + + _HEAD + + """ -@flow(name="engine", selectable=False) -def engine(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent.new()(task) - agent.new()(task) -''' +@flow(agents=Nothing, envs=Nowhere, params=FlowParams) +async def whatever_it_is_called(task, *, agents, envs, params, ctx): + return "one" +""" +) -#: A file with a `run` in it and nothing marked, which is what a flow used to be and is not. -UNMARKED = '''"""A file that says nothing about which of its functions is a flow.""" +#: A module of two visible flows, neither named for its directory. +TWO = ( + '"""Two, and neither is the directory\'s own."""\n' + + _HEAD + + ''' -from hmz.coganchor.agents import AgentBase +@flow(agents=Nothing, envs=Nowhere, params=FlowParams) +async def left(task, *, agents, envs, params, ctx): + """Goes left.""" + return "left" -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent.new()(task) +@flow(agents=Nothing, envs=Nowhere, params=FlowParams) +async def right(task, *, agents, envs, params, ctx): + """Goes right.""" + return "right" ''' +) -def _written(tmp_path: Path, source: str, name: str = "three") -> str: - """Writes a flow out as a flow is -- a directory -- and answers with its path.""" - return str(written(tmp_path, name, source)) - - -def test_a_file_says_which_flows_it_holds_and_what_each_one_does( - tmp_path: Path, -) -> None: - """Read off the decorator, which is where a flow says what it is.""" - said = held(_written(tmp_path, THREE)) - - assert [(one.name, one.about) for one in said] == [ - ("gen-idea", "Opens a loose idea into a draft."), - ("build", "builds it, under review"), - ] - - -def test_the_name_is_what_the_decorator_was_told_and_not_the_function_s( - tmp_path: Path, -) -> None: - """A name written down where a flow is run must not change under whoever renames it.""" - named = {one.name for one in held(_written(tmp_path, THREE))} - - assert named == {"gen-idea", "build"} # not `first_pass`, and not `start_it` - - -def test_a_function_called_run_is_not_a_flow_for_being_called_that( - tmp_path: Path, -) -> None: - """Which is the whole of the rule: a file says which of its functions is a flow.""" - where = _written(tmp_path, UNMARKED, "unmarked") - - assert held(where) == [] - with pytest.raises(NotAFlow, match="nothing in it is marked @flow"): - drives(where) - - -def test_a_file_marked_once_is_one_flow_under_its_own_name(tmp_path: Path) -> None: - """However the function it marked is spelled, which is nothing to do with the name.""" - (said,) = held(_written(tmp_path, ONE, "one")) - - assert said.name == "" - # Nothing said its own line, so the file's own first line is what it says. - assert said.about == "Just the one, and it says what it does here." - assert drives(_written(tmp_path, ONE, "one")) == ("",) - - -def test_the_file_s_own_flow_is_listed_first(tmp_path: Path) -> None: - """It is the one the file is named after, and a list that put it second would read wrong.""" - said = held(_written(tmp_path, BOTH, "both")) +@pytest.fixture +def mine(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + """This project's own flows directory, with the project as where things are run from.""" + monkeypatch.chdir(tmp_path) + return tmp_path / ".humanize" / "flows" - assert [one.name for one in said] == ["", "twice"] +def _local() -> list[tuple[str, str]]: + """What this project's own flows are listed as, and what each says it does.""" + return [(one.name, one.about) for one in found() if one.whose == LOCAL] -def test_each_flow_in_a_file_is_offered_under_its_own_name( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """`:`, which is what makes three of them three things to choose between.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - project = tmp_path / "project" - written(project / ".humanize/flows", "three", THREE) - monkeypatch.chdir(project) - listed = [(one.name, one.about) for one in found() if one.whose == "local"] +def test_a_module_lists_its_own_flow_bare_and_the_rest_by_name(mine: Path) -> None: + written(mine, "three", THREE) - assert listed == [ + assert _local() == [ + ("local/three", "builds it"), ("local/three:gen-idea", "Opens a loose idea into a draft."), - ("local/three:build", "builds it, under review"), - ] - - -def test_an_auxiliary_flow_is_callable_but_not_offered( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A composition engine is an API for another flow, not a choice for a person.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - project = tmp_path / "project" - at = written(project / ".humanize/flows", "composed", AUXILIARY) - monkeypatch.chdir(project) - - assert [(one.name, one.selectable) for one in held(at)] == [ - ("", True), - ("engine", False), ] - assert [one.name for one in found() if one.whose == "local"] == ["local/composed"] - assert drives("composed:engine") == ("",) - agent = ShellAgent(CONFIG) - Runner("composed:engine", [agent]).run("echo internal") - assert len(agent.opened) == 2 - - -def test_which_one_was_asked_for_is_the_half_after_the_colon(tmp_path: Path) -> None: - where = _written(tmp_path, THREE) - - assert inside(f"{where}:gen-idea") == "gen-idea" - assert inside(where) == "" # the one a flow holds under its own name - # The flow is the other half, and a path is still a path whatever is after it. - assert find(f"{where}:gen-idea") == f"{where}/{ENTRY}" - - -def test_each_of_them_asks_only_for_its_own_agents_and_settings( - tmp_path: Path, -) -> None: - """Which is the whole of what splitting one flow into three buys.""" - where = _written(tmp_path, THREE) - - assert drives(f"{where}:gen-idea") == ("drafter",) - assert drives(f"{where}:build") == ("builder", "reviewer") - idea = configures(f"{where}:gen-idea") - assert idea is not None - assert set(idea.model_fields) == {"n"} - assert configures(f"{where}:build") is None - - -def test_an_agent_a_flow_never_named_is_left_with_its_codename(tmp_path: Path) -> None: - """A flow that said how many it drives and no more has no name to give the one it gets.""" - where = _written(tmp_path, ONE, "one") - agent = ShellAgent(CONFIG) - drawn = agent.id - - Runner(where, [agent]).run("echo one") - - assert agent.id == drawn # the place had no name, so nothing renamed it - assert agent.id in SAID or re.fullmatch( # a designation out of Amphoreus - r"[A-Z][a-z]+(?:[A-Z][a-z]+)+[0-9]{3}", agent.id - ) -def test_the_one_that_was_asked_for_is_the_one_that_runs(tmp_path: Path) -> None: - where = _written(tmp_path, THREE) - builder, reviewer = ShellAgent(CONFIG), ShellAgent(CONFIG) +def test_a_module_of_one_flow_lists_it_under_the_directory(mine: Path) -> None: + """Whatever the function is called; and the module's docstring says what it does.""" + written(mine, "one", ONE) - Runner(f"{where}:build", [builder, reviewer]).run("echo built") + assert _local() == [("local/one", "Just the one, and it says what it does here.")] + assert resolved("one").name == "whatever_it_is_called" - assert builder.opened # the phase that was named, and no other - assert not reviewer.opened - -def test_a_file_of_several_asked_for_by_its_own_name_says_which_ones_it_holds( - tmp_path: Path, +def test_two_flows_neither_named_for_the_module_are_each_listed_by_name( + mine: Path, ) -> None: - """A colon away from what was meant, so the answer is the list of what to put after it.""" - where = _written(tmp_path, THREE) - - with pytest.raises(NotAFlow, match="three:gen-idea, three:build"): - drives(where) - - -def test_a_name_no_flow_in_the_file_answers_to_says_so(tmp_path: Path) -> None: - where = _written(tmp_path, THREE) - - with pytest.raises(NotAFlow, match="nothing in it is a flow called 'gen-plan'"): - drives(f"{where}:gen-plan") - + written(mine, "two", TWO) -def test_a_file_that_holds_one_may_still_be_asked_for_by_name(tmp_path: Path) -> None: - """The file's own flow has no name of its own, so a colon on it names nothing.""" - where = _written(tmp_path, ONE, "one") - - assert drives(where) == ("",) - with pytest.raises(NotAFlow, match="nothing in it is a flow called 'nope'"): - drives(f"{where}:nope") - - -def test_what_a_flow_says_about_itself_is_read_back_by_name(tmp_path: Path) -> None: - """Which is what a list of them shows beside each, so it is asked for by the same name.""" - where = _written(tmp_path, THREE) - - assert about(f"{where}:build") == "builds it, under review" - assert about(_written(tmp_path, ONE, "one")) == ( - "Just the one, and it says what it does here." - ) - - -def test_the_decorator_leaves_the_function_alone(tmp_path: Path) -> None: - """A flow is called the way it always was: what is added is what it says about itself.""" - del tmp_path - said: list[str] = [] - - @flow(about="what it does") - def two(one: str, other: str = "b") -> str: - said.append(one) - return one + other - - assert two("a") == "ab" - assert said == ["a"] - assert two.__name__ == "two" - - -def test_a_flow_a_file_holds_beside_its_own_is_reached_the_same_way( - tmp_path: Path, -) -> None: - """A file may be one flow and hold another, which is two names for two things.""" - where = _written(tmp_path, BOTH, "both") - - assert wanted(where) == wanted(f"{where}:twice") - agent = ShellAgent(CONFIG) - Runner(f"{where}:twice", [agent]).run("echo twice") - - assert ( - len(agent.opened) == 2 - ) # the one that opens two sessions, not the one that opens one - - -def test_a_flow_that_is_one_file_is_a_flow_too( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A flow is a module, and a single `.py` is one: it brings no skills, and runs.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - project = tmp_path / "project" - (project / ".humanize/flows").mkdir(parents=True) - (project / ".humanize/flows/alone.py").write_text(ONE) - monkeypatch.chdir(project) - - assert find("alone") == str((project / ".humanize/flows/alone.py").resolve()) - assert [one.name for one in found() if one.whose == "local"] == ["local/alone"] - assert drives("alone") == ("",) - agent = ShellAgent(CONFIG) - Runner("alone", [agent]).run("echo alone") - assert agent.opened - - -def test_a_flow_that_is_one_file_is_found_by_the_path_that_leaves_the_py_off( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Both shapes of a path, since a single-file flow is written down without its extension. - - A path is what is typed for a flow that is nowhere flows are kept, and it is also what - every older spelling of one of your own was. Either resolves, or a flow is findable one - way and not the other -- which is a flow that is offered and cannot be run. - """ - monkeypatch.setenv("HOME", str(tmp_path / "home")) - project = tmp_path / "project" - (project / ".humanize/flows").mkdir(parents=True) - (project / ".humanize/flows/alone.py").write_text(ONE) - monkeypatch.chdir(project) - at = str((project / ".humanize/flows/alone.py").resolve()) - - assert find(".humanize/flows/alone.py") == at # the file, pointed at outright - assert ( - find(".humanize/flows/alone") == at - ) # and the same path without the extension - - -def test_every_flow_that_is_offered_is_one_that_can_be_run( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """One rule names them and one rule finds them, so a listed name MUST resolve to a file. + assert _local() == [ + ("local/two:left", "Goes left."), + ("local/two:right", "Goes right."), + ] - The list and the lookup are the two halves that have to agree: a name that is offered and - then found nothing is a row in the picker that fails when it is chosen. - """ - monkeypatch.setenv("HOME", str(tmp_path / "home")) - project = tmp_path / "project" - written(project / ".humanize/flows", "whole", ONE) # a flow that is a directory - written(tmp_path / "home/.humanize/flows", "mine", ONE) # and one of yours - (project / ".humanize/flows/alone.py").write_text(ONE) # and one that is a file - (project / ".humanize/flows/three.py").write_text(THREE) # holding three of them - monkeypatch.chdir(project) - listed = found() +def test_a_bare_name_that_means_no_one_flow_says_which_there_are(mine: Path) -> None: + written(mine, "two", TWO) - assert {"local/whole", "local/alone", "user/mine", "local/three:build"} <= { - one.name for one in listed - } - assert [one.name for one in listed if not Path(find(one.name)).is_file()] == [] + with pytest.raises(FlowNotFound, match="left, right"): + resolved("two") + assert resolved("two:right").name == "right" -def test_a_directory_wins_a_name_a_file_also_uses( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch +def test_a_hidden_flow_is_listed_nowhere_and_still_answers_to_its_name( + mine: Path, ) -> None: - """The one that says most about itself: a flow with a `skills/` cannot be a file.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - project = tmp_path / "project" - written(project / ".humanize/flows", "both", ONE) - (project / ".humanize/flows/both.py").write_text(UNMARKED) - monkeypatch.chdir(project) + written(mine, "three", THREE) - assert find("both") == str((project / ".humanize/flows/both" / ENTRY).resolve()) - # And it is offered once rather than twice, under the one name it has. - assert [one.name for one in found() if one.whose == "local"] == ["local/both"] + assert all("engine" not in name for name, _ in _local()) + assert resolved("three:engine").hidden -#: A flow that reads what it does out of the module beside it, which is how a flow keeps a -#: prompt, a schedule or a table of its own without putting it in the flow itself. -BESIDE = '''"""Says what the module beside it says.""" +def test_which_one_was_asked_for_is_the_half_after_the_colon(mine: Path) -> None: + written(mine, "three", THREE) -import beside + assert resolved("three:gen-idea").name == "gen-idea" + assert resolved("local/three:gen-idea").name == "gen-idea" + assert resolved("three").name == "three" + assert about("three:gen-idea") == "Opens a loose idea into a draft." -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +def test_a_function_that_is_not_marked_is_not_a_flow(mine: Path) -> None: + written(mine, "three", THREE) -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent(f"echo {beside.SAYS} > said.txt") -''' + with pytest.raises(FlowNotFound, match="run"): + resolved("three:run") -def test_each_flow_reads_the_module_beside_it_rather_than_the_last_flows( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Every flow may have a `beside.py`, and one process may run several of them. +def test_a_flow_that_is_one_file_is_a_flow_too(mine: Path) -> None: + mine.mkdir(parents=True) + (mine / "alone.py").write_text(ONE) - A module imported by its plain name is cached under that plain name, so the first flow - loaded would own it: the second flow's `import beside` would be answered with the first - one's, and drawing the menu -- which loads every flow there is -- would settle which. - """ - monkeypatch.setenv("HOME", str(tmp_path / "home")) - project = tmp_path / "project" - for name in ("alpha", "beta"): - at = written(project / ".humanize/flows", name, BESIDE) - (at / "beside.py").write_text(f'SAYS = "{name}"\n') - monkeypatch.chdir(project) + assert _local() == [("local/alone", "Just the one, and it says what it does here.")] + assert find("alone") == str((mine / "alone.py").resolve()) + assert resolved("alone").name == "whatever_it_is_called" - held( - str(project / ".humanize/flows/alpha") - ) # as the menu does, to say what they are - Runner("beta", [ShellAgent(CONFIG)]).run("") - assert (project / "said.txt").read_text().strip() == "beta" +def test_a_directory_wins_a_name_a_file_also_uses(mine: Path) -> None: + written(mine, "both", ONE) + (mine / "both.py").write_text(TWO) + assert find("both") == str((mine / "both" / ENTRY).resolve()) + assert [name for name, _ in _local()] == ["local/both"] -def test_a_module_beside_a_flow_rewritten_between_runs_is_read_again( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Which is what a flow that improves itself does: the prompt beside it is where it is.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - project = tmp_path / "project" - at = written(project / ".humanize/flows", "mine", BESIDE) - (at / "beside.py").write_text('SAYS = "first"\n') - monkeypatch.chdir(project) - - Runner("mine", [ShellAgent(CONFIG)]).run("") - assert (project / "said.txt").read_text().strip() == "first" - (at / "beside.py").write_text('SAYS = "second"\n') - Runner("mine", [ShellAgent(CONFIG)]).run("") +def test_a_flow_is_found_by_its_path_as_well(tmp_path: Path) -> None: + """A flow anywhere else is its directory, or its file, typed out.""" + at = written(tmp_path / "elsewhere", "three", THREE) - assert (project / "said.txt").read_text().strip() == "second" + assert resolved(str(at)).name == "three" + assert resolved(f"{at}:gen-idea").name == "gen-idea" diff --git a/tests/integration/flows/test_flow_skills.py b/tests/integration/flows/test_flow_skills.py index 084073f0..c7095190 100644 --- a/tests/integration/flows/test_flow_skills.py +++ b/tests/integration/flows/test_flow_skills.py @@ -1,15 +1,21 @@ """The skills a flow brings, and the sessions they are mounted onto. -A flow is a directory, and `skills/` inside it is what that flow works by. Every session its -agents open is given them -- copied where that backend reads a project's own skills, for as -long as the session lives, and taken away again after -- so a flow carries what it needs -rather than expecting it to be installed on whoever's machine runs it. A flow may also name -skills that live in somebody else's repository, which are fetched and then mounted the same -way. +A flow is a directory, and `skills/` inside it is what that flow works by. A role names the +ones its agent is given -- `_skills` where the role is declared -- and every session that agent +opens is given them: copied where that backend reads a project's own skills, for as long as +the session lives, and taken away again after, so a flow carries what it needs rather than +expecting it to be installed on whoever's machine runs it. A role may also name skills that +live in somebody else's repository, which are fetched and then mounted the same way. + +Most of what is checked here is the mount, which is the agent's: what the runtime hands a +session's agent -- :func:`~hmz.runtime.flowing.skills.brought`, loaded with `loads` -- and what +the agent does with it. A run through the runtime, with a stand-in CLI, checks that a role is +given the skills it names and no others. """ from __future__ import annotations +import contextlib import gc import subprocess import time @@ -17,11 +23,11 @@ import pytest -from hmz._legacy_flows import NotAFlow -from hmz.coganchor.agents import AgentConfig +from hmz.coganchor.agents import AgentBase, AgentConfig from hmz.coganchor.agents.skills import Loaded from hmz.runtime.flowing.skills import brought, cached -from hmz.runtime.runner import Runner +from hmz.runtime.runner import Refused, Runner +from tests.flows import standins from tests.stubs import ShellAgent, written if TYPE_CHECKING: @@ -71,58 +77,74 @@ class PiAgent(ShellAgent): """A stand-in named for pi, whose project directories are read only once approved.""" -#: A flow that does what it is told, in a session of its own: what the turn is is the shell -#: line the test hands it, so a test can have it look at what was mounted beside it. +#: A flow that is a directory, which is what brings skills: what it does is not the point. DOES = '''"""Does the one thing it is told.""" -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent(task) -''' +class Agents(AgentCollection): + agent: Agent -#: The same, holding one session across two turns, so that two of them are open at once. -TWICE = '''"""Opens two sessions and does the thing in both.""" -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +class Envs(EnvCollection): + here: LocalEnv -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - one, other = agent.new(), agent.new() - one(task) - other(task) - one.close() - # The other is still open, so what they share is still there for it. - other("ls .claude/skills > while-one-is-shut.txt") - other.close() +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def mine(task, *, agents, envs, params, ctx): + session = await agents["agent"].spawn(env=envs["here"]) + return await agents["agent"].run(task, session=session) ''' -#: A flow that holds a session open and calls another flow while it is open, which is what -#: `load` is for: the two of them are running at once, in one workspace. -CALLING = '''"""Holds a session open, then calls another flow.""" +#: A flow whose role names one of its skills, and reads what its session was given while the +#: session is open. NAMED is filled in per test. +READS = '''"""Reads what its agent's session was given.""" + +from hmz.flows import ( + Agent, + AgentCollection, + EnvCollection, + FilesEnvMixin, + FlowParams, + LocalEnv, + ShellEnvMixin, + flow, +) + -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow -from hmz._legacy_flows import load +class Noter(Agent): + _skills = NAMED -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - held = agent.new() - held("true") # a turn, so this flow's session is open and its skills are mounted - load("inner")(agents, task) - held.close() +class Here(LocalEnv, FilesEnvMixin, ShellEnvMixin): ... + + +class Agents(AgentCollection): + agent: Noter + + +class Envs(EnvCollection): + here: Here + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def reads(task, *, agents, envs, params, ctx): + here = envs["here"] + session = await agents["agent"].spawn(env=here) + await agents["agent"].run(task, session=session) + code, out, _ = await here.exec(["ls", ".claude/skills"]) + said = await here.read(".claude/skills/note-taking/SKILL.md") if "note-taking" in out else b"" + return out.split(), said.decode() ''' +def _given(agent: AgentBase, flow: Path, *urls: str) -> AgentBase: + """What the runtime hands a session's agent: the skills the flow brings, loaded onto it.""" + agent.loads(brought(flow, urls)) + return agent + + def skill(name: str, says: str = "Do the thing.") -> str: """One `SKILL.md`, as every one of these CLIs lays one out.""" return f"---\nname: {name}\ndescription: does a thing\n---\n\n{says}\n" @@ -221,18 +243,21 @@ def test_the_flows_own_wins_a_name_a_repository_also_uses(tmp_path: Path) -> Non def test_a_repository_that_cannot_be_reached_stops_the_run_before_it_starts( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, claude: None ) -> None: """A flow that works by a skill it has not got is not one to start and find out later.""" monkeypatch.chdir(tmp_path) written( tmp_path / ".humanize/flows", - "mine", - DOES.replace("@flow\n", '@flow(skills=("/nowhere/at/all",))\n'), + "reads", + READS.replace("NAMED", '("/nowhere/at/all#note-taking",)'), + ) + run = Runner( + "reads", agents={"agent": "claude/claude-haiku-4-5:low"}, budget={"cost": 1} ) - with pytest.raises(NotAFlow): - Runner("mine", [ClaudeAgent(CONFIG)]) + with pytest.raises(Refused, match="cannot be fetched"): + run.run("Reply with the single word: hi") def test_a_session_is_given_them_where_its_backend_reads_a_projects_own( @@ -240,14 +265,14 @@ def test_a_session_is_given_them_where_its_backend_reads_a_projects_own( ) -> None: """Mounted as the session opens, and gone once the session it was for has ended.""" monkeypatch.chdir(tmp_path) - written( + at = written( tmp_path / ".humanize/flows", "mine", DOES, {"note-taking": skill("note-taking")}, ) - Runner("mine", [ClaudeAgent(CONFIG)]).run( + _given(ClaudeAgent(CONFIG), at)( "ls .claude/skills > listed.txt; cat .claude/skills/note-taking/SKILL.md > read.txt" ) gc.collect() # the session the turn ran in, let go of by the flow @@ -277,14 +302,14 @@ def test_a_shared_skill_backend_is_given_flow_skills_in_the_project_directory( ) -> None: """Every backend that reads the shared project directory is given the flow's skills.""" monkeypatch.chdir(tmp_path) - written( + at = written( tmp_path / ".humanize/flows", "mine", DOES, {"note-taking": skill("note-taking")}, ) - Runner("mine", [agent_type(CONFIG)]).run( + _given(agent_type(CONFIG), at)( "ls .agents/skills > listed.txt; " "cat .agents/skills/note-taking/SKILL.md > read.txt" ) @@ -305,14 +330,14 @@ def test_a_backend_that_would_not_read_them_there_is_given_none( leave a directory in somebody's project that no turn of that flow would ever read. """ monkeypatch.chdir(tmp_path) - written( + at = written( tmp_path / ".humanize/flows", "mine", DOES, {"note-taking": skill("note-taking")}, ) - Runner("mine", [PiAgent(CONFIG)]).run("ls -a > listed.txt") + _given(PiAgent(CONFIG), at)("ls -a > listed.txt") gc.collect() assert ".agents" not in (tmp_path / "listed.txt").read_text().split() @@ -328,14 +353,14 @@ def test_a_projects_own_skill_of_that_name_is_left_alone( (tmp_path / ".claude" / "skills" / "note-taking" / "SKILL.md").write_text( skill("note-taking", says="The project's own.") ) - written( + at = written( tmp_path / ".humanize/flows", "mine", DOES, {"note-taking": skill("note-taking")}, ) - Runner("mine", [ClaudeAgent(CONFIG)]).run( + _given(ClaudeAgent(CONFIG), at)( "cat .claude/skills/note-taking/SKILL.md > read.txt" ) gc.collect() @@ -350,14 +375,16 @@ def test_two_sessions_of_one_flow_share_the_mount_until_the_last_is_done( ) -> None: """One closing must not take out from under the other what both were given.""" monkeypatch.chdir(tmp_path) - written( - tmp_path / ".humanize/flows", - "twice", - TWICE, - {"note-taking": skill("note-taking")}, - ) + at = written(tmp_path, "twice", DOES, {"note-taking": skill("note-taking")}) + agent = _given(ClaudeAgent(CONFIG), at) - Runner("twice", [ClaudeAgent(CONFIG)]).run("ls .claude/skills > listed.txt") + one, other = agent.new(), agent.new() + one("true") + other("true") + one.close() + # The other is still open, so what they share is still there for it. + other("ls .claude/skills > while-one-is-shut.txt") + other.close() gc.collect() assert (tmp_path / "while-one-is-shut.txt").read_text().split() == ["note-taking"] @@ -372,25 +399,31 @@ def test_a_called_flows_skill_does_not_take_over_the_name_from_the_flow_that_cal A mount is one directory per name, shared between the sessions holding it. Shared by name alone, a flow calling another flow whose skills happen to be named the same would have the called flow's session reading the caller's skill -- and, worse, the caller's session - reading the called one's after it ended and the count fell. + reading the called one's after it ended and the count fell. Each flow's sessions are + their own agent's, given that flow's skills. """ monkeypatch.chdir(tmp_path) - written( - tmp_path / ".humanize/flows", + outer = written( + tmp_path / "flows", "outer", - CALLING, + DOES, {"note-taking": skill("note-taking", says="The outer flow's.")}, ) - written( - tmp_path / ".humanize/flows", + inner = written( + tmp_path / "flows", "inner", DOES, {"note-taking": skill("note-taking", says="The inner flow's.")}, ) - Runner("outer", [ClaudeAgent(CONFIG)]).run( + held = _given(ClaudeAgent(CONFIG), outer).new() + held( + "true" + ) # a turn, so the outer flow's session is open and its skills are mounted + _given(ClaudeAgent(CONFIG), inner)( "cat .claude/skills/note-taking/SKILL.md > read.txt" ) + held.close() gc.collect() # The one that was there first is the one both read: a name is one skill to the CLI, and @@ -399,59 +432,18 @@ def test_a_called_flows_skill_does_not_take_over_the_name_from_the_flow_that_cal assert not (tmp_path / ".claude").exists() # and both of them are taken away after -#: A flow that closes its session and then goes on working in another, which is what a flow -#: whose agent was stopped and started again does. -AGAIN = '''"""Closes a session and opens another.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - one = agent.new() - one("true") - one.close() - agent.new()(task) -''' - -#: A flow whose turn takes long enough to be stopped in the middle of, and reads the skill it -#: was given after it has been. -SLOWLY = '''"""Takes one long turn.""" - -import threading - -from hmz.coganchor.agents import AgentBase, Stopped -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - threading.Timer(0.2, agent.stop).start() - try: - agent(task) - except Stopped: - pass -''' - - def test_a_session_opened_after_one_was_closed_is_given_them_too( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """What a closing takes away is the closing session's, not the flow's for the rest of it.""" monkeypatch.chdir(tmp_path) - written( - tmp_path / ".humanize/flows", - "again", - AGAIN, - {"note-taking": skill("note-taking")}, - ) + at = written(tmp_path, "again", DOES, {"note-taking": skill("note-taking")}) + agent = _given(ClaudeAgent(CONFIG), at) - Runner("again", [ClaudeAgent(CONFIG)]).run( - "cat .claude/skills/note-taking/SKILL.md > read.txt" - ) + one = agent.new() + one("true") + one.close() + agent.new()("cat .claude/skills/note-taking/SKILL.md > read.txt") gc.collect() assert "Do the thing" in (tmp_path / "read.txt").read_text() @@ -470,18 +462,18 @@ def test_a_stop_ends_the_turns_process_and_takes_the_skills_after_it( stopping it, only breaking it. So the turn lets go of them as it unwinds, which is the first moment nothing is working by them. """ + import threading + + from hmz.coganchor.agents import Stopped + monkeypatch.chdir(tmp_path) - written( - tmp_path / ".humanize/flows", - "slowly", - SLOWLY, - {"note-taking": skill("note-taking")}, - ) + at = written(tmp_path, "slowly", DOES, {"note-taking": skill("note-taking")}) + agent = _given(ClaudeAgent(CONFIG), at) began = time.monotonic() - Runner("slowly", [ClaudeAgent(CONFIG)]).run( - "cat .claude/skills/note-taking/SKILL.md > read.txt; sleep 30" - ) + threading.Timer(0.2, agent.stop).start() + with contextlib.suppress(Stopped): + agent("cat .claude/skills/note-taking/SKILL.md > read.txt; sleep 30") took = time.monotonic() - began gc.collect() @@ -498,14 +490,14 @@ def test_a_directory_the_project_already_had_is_left_where_it_is( """Even an empty one: what humanize takes away is what humanize made.""" monkeypatch.chdir(tmp_path) (tmp_path / ".claude" / "skills").mkdir(parents=True) - written( + at = written( tmp_path / ".humanize/flows", "mine", DOES, {"note-taking": skill("note-taking")}, ) - Runner("mine", [ClaudeAgent(CONFIG)]).run("true") + _given(ClaudeAgent(CONFIG), at)("true") gc.collect() assert (tmp_path / ".claude" / "skills").is_dir() @@ -545,9 +537,10 @@ def test_a_backend_that_reads_no_such_directory_carries_none( DOES, {"note-taking": skill("note-taking")}, ) - agent = ShellAgent(CONFIG) # a backend with nowhere a skill of a flow's could go + # A backend with nowhere a skill of a flow's could go. + agent = _given(ShellAgent(CONFIG), tmp_path / ".humanize/flows/mine") - Runner("mine", [agent]).run("ls -a > listed.txt") + agent("ls -a > listed.txt") assert agent.loaded == ( Loaded( @@ -630,7 +623,9 @@ def test_a_copy_is_yours_to_change_and_is_what_then_runs( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """A fetched flowverse is fetched over, so an edit that is to keep is an edit to a copy.""" - from hmz.runtime.flowing import fork + from pathlib import Path + + from hmz.runtime.flowing import find, fork monkeypatch.chdir(tmp_path) (tmp_path / "theirs").mkdir() @@ -639,9 +634,60 @@ def test_a_copy_is_yours_to_change_and_is_what_then_runs( at = tmp_path / ".humanize/flows/mine/skills/note-taking/SKILL.md" at.write_text(skill("note-taking", says="Changed, and mine.")) - Runner("mine", [ClaudeAgent(CONFIG)]).run( + _given(ClaudeAgent(CONFIG), Path(find("mine")).parent)( "cat .claude/skills/note-taking/SKILL.md > read.txt" ) gc.collect() assert "Changed, and mine" in (tmp_path / "read.txt").read_text() + + +@pytest.fixture +def claude(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """A stand-in Claude Code on PATH, with a home of its own.""" + standins.install(tmp_path / "bin", "claude", standins.CLAUDE) + monkeypatch.setenv("PATH", standins.path_with(tmp_path / "bin")) + monkeypatch.setenv("HOME", str(tmp_path / "home")) + monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(tmp_path / "claude-home")) + + +def test_a_role_is_given_the_skills_it_names_through_a_run( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, claude: None +) -> None: + """What the runtime hands a session: the role's own skills, mounted while it is open.""" + monkeypatch.chdir(tmp_path) + written( + tmp_path / ".humanize/flows", + "reads", + READS.replace("NAMED", '("note-taking",)'), + {"note-taking": skill("note-taking"), "other": skill("other")}, + ) + + listed, said = Runner( + "reads", + agents={"agent": "claude/claude-haiku-4-5:low"}, + budget={"cost": 1}, + ).run("Reply with the single word: hi") + + # The one the role named, and not the flow's other skill. + assert listed == ["note-taking"] + assert "does a thing" in said + assert not (tmp_path / ".claude").exists() # and gone once the run is over + + +def test_a_role_naming_a_skill_the_flow_has_not_got_stops_the_run( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, claude: None +) -> None: + monkeypatch.chdir(tmp_path) + written( + tmp_path / ".humanize/flows", + "reads", + READS.replace("NAMED", '("nowhere",)'), + {"note-taking": skill("note-taking")}, + ) + run = Runner( + "reads", agents={"agent": "claude/claude-haiku-4-5:low"}, budget={"cost": 1} + ) + + with pytest.raises(Refused, match="nowhere"): + run.run("Reply with the single word: hi") diff --git a/tests/integration/flows/test_flowverses.py b/tests/integration/flows/test_flowverses.py index 2dbc10a9..7abb03f4 100644 --- a/tests/integration/flows/test_flowverses.py +++ b/tests/integration/flows/test_flowverses.py @@ -12,12 +12,14 @@ from __future__ import annotations +import asyncio import subprocess import threading from typing import TYPE_CHECKING import pytest +from hmz.flows import FlowNotFound from hmz.runtime.flowing import ( BUILTIN_AT, ENTRY, @@ -28,6 +30,7 @@ find, flowverses, found, + resolved, ) from hmz.runtime.flowing import verses as store from tests.stubs import written @@ -38,14 +41,21 @@ #: A flow, as short as one can be: the file is what is being fetched, not what it does. FLOW = '''"""A flow of somebody else's.""" -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent.new()(task) +class Agents(AgentCollection): + agent: Agent + + +class Envs(EnvCollection): + here: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def run(task, *, agents, envs, params, ctx): + session = await agents["agent"].spawn(env=envs["here"]) + return await agents["agent"].run(task, session=session) ''' @@ -494,22 +504,23 @@ def test_a_flow_of_your_own_still_wins_a_bare_name( def test_a_flow_from_a_flowverse_runs_by_that_name(theirs: Path) -> None: """Which is the whole point of fetching one: `-f theirs/loop` is a flow to run.""" - from hmz.runtime.flowing import drives + from hmz.runtime.flowing.fakes import FakeAgentDriver, run_fake store.add(str(theirs)) + flow = resolved("theirs/loop") - assert drives("theirs/loop") == ("",) # one agent, and the flow calls it nothing + assert [one.name for one in flow.describe().agents] == ["agent"] + assert ( + asyncio.run(run_fake(flow, "hi", agents={"agent": FakeAgentDriver()})) == "ok" + ) def test_a_flowverse_that_has_not_been_fetched_says_so_rather_than_that_there_is_no_file() -> ( None ): """The name is right and the download has not happened, which is a different thing.""" - from hmz._legacy_flows import NotAFlow - from hmz.runtime.flowing import drives - - with pytest.raises(NotAFlow, match="has not been fetched yet"): - drives(f"{OFFICIAL}/rlar") + with pytest.raises(FlowNotFound, match="has not been fetched yet"): + resolved(f"{OFFICIAL}/rlar") def test_a_bare_name_says_so_too_when_nothing_has_been_fetched(theirs: Path) -> None: @@ -518,15 +529,12 @@ def test_a_bare_name_says_so_too_when_nothing_has_been_fetched(theirs: Path) -> `-f rlar` on a machine that has fetched nothing is a name that is right and a download that has not happened, which "no flow to read" is the least useful thing to say about. """ - from hmz._legacy_flows import NotAFlow - from hmz.runtime.flowing import drives - store.add(str(theirs)) # one that is here, so the one that is not is named alone with pytest.raises( - NotAFlow, match=f"the {OFFICIAL} flowverse has not been fetched yet" + FlowNotFound, match=f"the {OFFICIAL} flowverse has not been fetched yet" ): - drives("rlar") + resolved("rlar") def test_a_clone_somebody_has_written_into_says_so(theirs: Path) -> None: diff --git a/tests/integration/flows/test_proving.py b/tests/integration/flows/test_proving.py deleted file mode 100644 index 84acc9ae..00000000 --- a/tests/integration/flows/test_proving.py +++ /dev/null @@ -1,515 +0,0 @@ -"""A flow driven by stubs against a clock, held to ending every proof and saying why. - -The scenarios are the questions: a budget loop walks to the end of its budget when the -reviewer never says done, a loop whose only exit is that verdict is caught by the turn cap, -a flow that takes no turns at all is killed by the clock, and the silent world -- every turn -answering nothing -- is every guard tried at once. Each proof is a subprocess, so these -tests are also the test that the child half answers the parent at all. -""" - -from __future__ import annotations - -import textwrap -from typing import TYPE_CHECKING, Literal - -from pydantic import BaseModel, Field - -from hmz.runtime.flowing.proving import ( - ALWAYS_DONE, - NEVER_DONE, - SILENT, - Scenario, - _made, - _said, - proved, -) -from tests.stubs import written - -if TYPE_CHECKING: - from pathlib import Path - -#: A world that says no quickly: few turns allowed, and a short clock, so a flow that -#: cannot end is caught in test time rather than in a minute apiece. -QUICKLY = Scenario( - "never-done", verdict=False, answer="did some of it", turns=6, seconds=20.0 -) - -#: A budget loop in miniature: each stub turn climbs 100k output tokens, so three turns -#: spend the 250k this flow holds itself to. -BUDGETED = ''' -"""A loop held to a budget of its own.""" - -import time - -from hmz._legacy_flows import Agent, flow - - -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - while True: - agent(task, suppress=True) - if agent.spent().output >= 250_000: - return - time.sleep(5) -''' - -#: A loop whose one way out is the reviewer's verdict, which is rlar's shape. -VERDICT_ONLY = ''' -"""A loop only its reviewer can end.""" - -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - done: bool = Field(description="whether it is over") - - -@flow -def run(agents: tuple[Agent, Agent], task: str) -> None: - working = agents[0].new() - while True: - working(task, suppress=True) - review = agents[1](task, suppress=True, schema=Review) - if review is not None and review.done: - return -''' - - -def test_a_budget_loop_ends_when_the_reviewer_never_says_done(tmp_path: Path) -> None: - """The point of the whole module: bounded flows end in the worst world there is.""" - at = written(tmp_path, "one", textwrap.dedent(BUDGETED)) - proof = proved(at, scenarios=(NEVER_DONE,)) - assert proof.findings == () - (outcome,) = proof.outcomes - assert outcome.finished - assert outcome.turns == 3 - assert outcome.said == "" - - -def test_a_verdict_only_loop_is_caught_by_the_turn_cap(tmp_path: Path) -> None: - at = written(tmp_path, "one", textwrap.dedent(VERDICT_ONLY)) - proof = proved(at, scenarios=(QUICKLY, ALWAYS_DONE)) - never, always = proof.outcomes - assert not never.finished - assert never.turns == QUICKLY.turns - assert f"{QUICKLY.turns} turns" in never.said - # The same loop under a reviewer that says yes ends the way it means to. - assert always.finished - assert always.turns == 2 - - -def test_a_flow_that_takes_no_turns_is_killed_by_the_clock(tmp_path: Path) -> None: - at = written( - tmp_path, - "one", - textwrap.dedent( - ''' - """A loop with no turns for the cap to count.""" - - from hmz._legacy_flows import Agent, flow - - - @flow - def run(agents: tuple[Agent], task: str) -> None: - held = 0 - while True: - held += 1 - ''' - ), - ) - proof = proved(at, scenarios=(Scenario("spin", None, "", seconds=3.0),)) - (outcome,) = proof.outcomes - assert not outcome.finished - assert "still running after 3s" in outcome.said - - -def test_a_crash_is_the_outcome_with_its_last_words(tmp_path: Path) -> None: - at = written( - tmp_path, - "one", - textwrap.dedent( - ''' - """A flow that falls over.""" - - from hmz._legacy_flows import Agent, flow - - - @flow - def run(agents: tuple[Agent], task: str) -> None: - raise RuntimeError("the kettle is on fire") - ''' - ), - ) - proof = proved(at, scenarios=(ALWAYS_DONE,)) - (outcome,) = proof.outcomes - assert not outcome.finished - assert "the kettle is on fire" in outcome.said - - -def test_what_is_not_a_flow_is_a_refused_load(tmp_path: Path) -> None: - at = written(tmp_path, "one", '"""Not a flow."""\n\nx = 1\n') - proof = proved(at, scenarios=(ALWAYS_DONE,)) - assert [one.code for one in proof.findings] == ["refused-load"] - assert proof.findings[0].severity == "error" - assert "marked @flow()" in proof.findings[0].said - (outcome,) = proof.outcomes - assert not outcome.finished - - -#: A flow whose bound is its config, for proving the config reaches it. -CONFIGURED = ''' -"""A loop held to whatever budget it is set up with.""" - -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - - -class Config(BaseModel): - model_config = {"extra": "forbid"} - - budget: float = Field(default=1.0, ge=0, description="millions of output tokens") - - -@flow -def run(agents: tuple[Agent], task: str, config: Config | None = None) -> None: - (agent,) = agents - held = config or Config() - while True: - agent(task, suppress=True) - if agent.spent().output >= held.budget * 1_000_000: - return -''' - - -def test_a_config_is_read_back_through_the_flows_own_model(tmp_path: Path) -> None: - at = written(tmp_path, "one", textwrap.dedent(CONFIGURED)) - # 0.3 million at 100k a turn is three turns: the setting reached the loop. - proof = proved(at, config={"budget": 0.3}, scenarios=(NEVER_DONE,)) - assert proof.findings == () - assert proof.outcomes[0].finished - assert proof.outcomes[0].turns == 3 - # And one the model refuses is refused before anything runs. - refused = proved(at, config={"budget": "a lot"}, scenarios=(NEVER_DONE,)) - assert [one.code for one in refused.findings] == ["refused-load"] - assert not refused.outcomes[0].finished - - -#: A loop with no exit of its own at all, which is what a flow looks like once it stops -#: implementing a budget: what ends it is the allowance it declares. -DECLARED = ''' -"""A loop that ends when the allowance the flow declared is spent.""" - -from hmz._legacy_flows import Agent, Allowance, flow - - -@flow(budget=Allowance(tokens=0.3)) -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - while True: - agent(task, suppress=True) -''' - -#: The same loop saying nothing about what a run of it is worth, which is a flow with -#: nothing inside it to end it at all. -UNDECLARED = DECLARED.replace("@flow(budget=Allowance(tokens=0.3))", "@flow").replace( - ", Allowance,", "," -) - - -def test_the_allowance_a_flow_declares_is_what_ends_its_proof(tmp_path: Path) -> None: - """So that a flow holding itself to nothing is still proved to end somewhere. - - Which is the whole point of the allowance being the run's rather than each flow's: the - loop below has no `break`, no `return` and no cap, and it ends. - """ - at = written(tmp_path, "declared", textwrap.dedent(DECLARED)) - - proof = proved(at, scenarios=(NEVER_DONE,)) - - assert proof.findings == () - assert proof.outcomes[0].finished - # 0.3 million at 100k a turn is three turns, as it would be for a cap the flow read off - # `spent()` itself -- which is what this replaces. - assert proof.outcomes[0].turns == 3 - - -#: The same loop, declaring its allowance in hours rather than in tokens -- which is the shape -#: a proof cannot measure unless its world has a clock that moves. -#: Deliberately not a whole number of ticks. The proof's clock is `began + turns * tick`, -#: so an allowance of exactly three minutes is reached at the very instant the third turn -#: is counted -- and whether the ledger reads the clock just before or just after that -#: counter moves is a race it loses about one run in ten on a loaded machine, which is a -#: test that fails saying `4 == 3` about nothing. 0.06 hours is 216 seconds, which no -#: multiple of the minute a turn is worth lands on: the third turn is 36 seconds short of -#: it and the fourth is 24 seconds past, so the same turn ends the walk either way round. -HOURS = DECLARED.replace("Allowance(tokens=0.3)", "Allowance(hours=0.06)") - - -def test_an_allowance_in_hours_is_walked_to_the_end_of_too(tmp_path: Path) -> None: - """A proof sleeps for free and a stub answers at once, so real time never moves in one. - - Left at that, a flow claiming six hours over a loop with no exit would be driven to the - turn cap and reported as a flow that does not stop -- a false failure about exactly the - shape the checker blesses. So a turn is worth a slice of clock, and the claim can be tried. - """ - at = written(tmp_path, "hours", textwrap.dedent(HOURS)) - - proof = proved(at, scenarios=(NEVER_DONE,)) - - assert proof.outcomes[0].finished - # 216 seconds, and a turn is worth the scenario's minute: the fourth reaches it. - assert proof.outcomes[0].turns == 4 - - -def test_a_flow_that_stopped_itself_did_not_reach_what_it_declared( - tmp_path: Path, -) -> None: - """`Stopped` is also what a flow that stops its own agent gets on its next turn. - - Read as the allowance being reached, such a flow would be filed as one that ended the way - its author said it should -- when what happened is that it stopped itself by mistake. - """ - at = written( - tmp_path, - "own", - textwrap.dedent( - ''' - """A loop that stops its own agent and then asks it for another turn.""" - - from hmz._legacy_flows import Agent, Allowance, flow - - - @flow(budget=Allowance(tokens=100.0)) - def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - while True: - agent(task) - agent.stop() - ''' - ), - ) - - proof = proved(at, scenarios=(NEVER_DONE,)) - - assert not proof.outcomes[0].finished - assert "without having run out of what it declared" in proof.outcomes[0].said - - -def test_a_flow_that_declares_nothing_is_held_to_nothing_here(tmp_path: Path) -> None: - """A proof of the flow does not stand on an allowance the flow never claimed. - - The one a real run has is whoever started it's rather than the flow's, and a loop - passing because somebody's money ran out is a loop nobody proved anything about. - """ - at = written(tmp_path, "undeclared", textwrap.dedent(UNDECLARED)) - - proof = proved(at, scenarios=(NEVER_DONE,)) - - assert not proof.outcomes[0].finished - assert "still going" in proof.outcomes[0].said - - -def test_an_empty_proof_only_loads_and_reads_the_live_config(tmp_path: Path) -> None: - at = written(tmp_path, "one", textwrap.dedent(CONFIGURED)) - proof = proved(at, scenarios=()) - assert proof == ((), ()) - # A config the static reading cannot see is still read here, off the model itself. - loose = written( - tmp_path, - "loose", - textwrap.dedent( - ''' - """A flow with a config that says nothing about itself.""" - - from hmz._legacy_flows import Agent, flow - from pydantic import BaseModel - - - class Config(BaseModel): - budget: float = 1.0 - - - @flow - def run(agents: tuple[Agent], task: str, config: Config | None = None) -> None: - agents[0](task) - ''' - ), - ) - told = proved(loose, scenarios=()) - assert [one.code for one in told.findings] == ["loose-config", "unsaid-field"] - assert {one.severity for one in told.findings} == {"warning"} - - -#: A flow reading a shaped answer: one guarded, one not, for the silent world to tell apart. -GUARDED = ''' -"""A flow that guards what a turn answered.""" - -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - done: bool = Field(description="whether it is over") - - -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - while True: - review = agent(task, suppress=True, schema=Review) - if review is not None and review.done: - return - if agent.spent().output >= 200_000: - return -''' - -UNGUARDED = ''' -"""A flow that reads a field off whatever came back.""" - -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - done: bool = Field(description="whether it is over") - - -@flow -def run(agents: tuple[Agent], task: str) -> None: - review = agents[0](task, suppress=True, schema=Review) - print(review.done) -''' - - -def test_the_silent_world_is_every_guard_tried_at_once(tmp_path: Path) -> None: - guarded = written(tmp_path, "guarded", textwrap.dedent(GUARDED)) - proof = proved(guarded, scenarios=(SILENT,)) - assert proof.outcomes[0].finished # None answers taken again, until the budget - unguarded = written(tmp_path, "unguarded", textwrap.dedent(UNGUARDED)) - told = proved(unguarded, scenarios=(SILENT,)) - assert not told.outcomes[0].finished - assert "AttributeError" in told.outcomes[0].said - - -def test_the_person_answers_what_the_scenario_says(tmp_path: Path) -> None: - at = written( - tmp_path, - "one", - textwrap.dedent( - ''' - """A conversation, over when the person says nothing.""" - - from typing import NamedTuple - - from hmz._legacy_flows import Agent, Person, flow - - - class Chat(NamedTuple): - assistant: Agent - human: Person - - - @flow - def run(agents: Chat, task: str) -> None: - conversation = agents.assistant.new() - said = task - while said: - answered = conversation(said, suppress=True) - said = agents.human(answered) - ''' - ), - ) - # Nobody at the prompt: the flow does the one thing it was given and stops. - silent = proved(at, scenarios=(SILENT,)) - assert silent.outcomes[0].finished - assert silent.outcomes[0].turns == 2 # the assistant's turn, and the person's "" - # A person who never stops talking is a conversation that never ends: the cap's. - chatty = proved(at, scenarios=(Scenario("chatty", True, "go on", turns=9),)) - assert not chatty.outcomes[0].finished - - -def test_an_async_flow_is_awaited(tmp_path: Path) -> None: - at = written( - tmp_path, - "one", - textwrap.dedent( - ''' - """A flow written as a coroutine.""" - - from hmz._legacy_flows import Agent, flow - - - @flow - async def run(agents: tuple[Agent], task: str) -> None: - await agents[0].aturn(task, suppress=True) - ''' - ), - ) - proof = proved(at, scenarios=(ALWAYS_DONE,)) - (outcome,) = proof.outcomes - assert outcome.finished, outcome - assert outcome.turns == 1 - - -class Nested(BaseModel): - said: str - fine: bool - - -class Shaped(BaseModel): - done: bool - notes: str - stage: Literal["draft", "final"] - rounds: int = 4 - weight: float - parts: list[str] - inner: Nested - extra: str | None = None - - -def test_a_shaped_answer_is_fabricated_field_by_field() -> None: - made = _made(Shaped, NEVER_DONE) - held = Shaped.model_validate(made) - assert held.done is False # the verdict, whatever the field is called - assert held.notes == NEVER_DONE.answer - assert held.stage == "draft" # the first of a literal's few - assert held.rounds == 4 # the default, where the field has one - assert held.weight == 0 - assert held.parts == [] - assert held.inner == Nested(said=NEVER_DONE.answer, fine=False) - assert held.extra == NEVER_DONE.answer # a string, even behind a union - - -def test_a_shape_nothing_can_be_fabricated_for_answers_nothing() -> None: - class Impossible(BaseModel): - count: int = Field(ge=5) # fabricated as 0, which the model then refuses - - assert _said(Impossible, ALWAYS_DONE) == "" - # And the silent world answers nothing whatever the shape. - assert _said(Shaped, SILENT) == "" - assert _said(None, SILENT) == "" - assert _said(None, ALWAYS_DONE) == ALWAYS_DONE.answer - - -def test_a_list_that_takes_at_least_some_is_answered_with_that_many() -> None: - from typing import Annotated - - class Planned(BaseModel): - lanes: list[Nested] = Field(min_length=3) - tags: list[Annotated[str, Field(min_length=1)]] = Field(min_length=1) - loose: list[str] - - made = _made(Planned, NEVER_DONE) - held = Planned.model_validate(made) - assert len(held.lanes) == 3 - assert held.lanes[0] == Nested(said=NEVER_DONE.answer, fine=False) - assert held.tags == [NEVER_DONE.answer] - assert held.loose == [] diff --git a/tests/integration/flows/test_recursive_flows.py b/tests/integration/flows/test_recursive_flows.py deleted file mode 100644 index f7ad0bea..00000000 --- a/tests/integration/flows/test_recursive_flows.py +++ /dev/null @@ -1,590 +0,0 @@ -"""A flow that calls a flow that calls a flow, as deep as it likes and several at once. - -One flow reaching for another is one thing; a flow that decides how deep to go and runs two -branches of itself at the same time is the thing that has to be tracked rather than listed. A -run of those is a tree -- each flow under the one that called it, two gathered siblings under -neither -- and what is checked here is that it runs as one, is written down as one, reads back -as one, and unwinds as one when it is stopped partway. -""" - -from __future__ import annotations - -import asyncio -import json -from typing import TYPE_CHECKING, Any - -import pytest - -from hmz._legacy_flows import NotAFlow, load, running -from hmz.coganchor.agents import AgentConfig -from hmz.coganchor.agents.skills import Loaded -from hmz.runtime.epic import epics, records, sessions, tree -from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent, events, written - -if TYPE_CHECKING: - from pathlib import Path - -CONFIG = AgentConfig(model="m", effort="high") - -#: A flow that calls two of itself at once until it has gone as deep as it was told to, and -#: writes down the branch it was on at every level. Dynamic in the depth, since a recursion -#: whose depth is written into the file is a recursion a test wrote rather than one a flow did. -DEEPER = '''"""Two of itself at once, until it has gone deep enough.""" - -import asyncio -import json -from pathlib import Path - -from pydantic import BaseModel - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow, load, running - - -class Config(BaseModel): - """How much further to go.""" - - left: int = 0 - trail: str = "" - - -@flow -async def run(agents: tuple[AgentBase], task: str, config: Config | None = None) -> None: - setting = config or Config() - with Path("branches.jsonl").open("a") as out: - out.write(json.dumps({ - "trail": setting.trail, - "running": [one.flow for one in running()], - "deep": [one.depth for one in running()], - }) + "\\n") - if setting.left <= 0: - agents[0].new()("echo bottom") - return - # An agent apiece, which is what makes two branches of one run two branches: an agent - # both of them were driving would be writing where they were both called from. - await asyncio.gather(*( - load("deeper")( - [agents[0].clone()], - task, - {"left": setting.left - 1, "trail": setting.trail + side}, - ) - for side in ("L", "R") - )) -''' - -#: A flow that opens a session with its sibling holding the same agent, and does not leave -#: until the sibling's session is open too. Which makes "while both of them have it" a stretch -#: of time rather than a race: what an agent is writing to is read as each session opens. -OPENS = '''"""Opens a session while its sibling is running, and stays until that one has too.""" - -import asyncio -import uuid -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - -MINE = uuid.uuid4().hex - - -async def _both(named: str) -> None: - """Waits until this flow and the one gathered beside it have both got this far.""" - Path(f"{named}-{MINE}.txt").write_text("") - while len(list(Path.cwd().glob(f"{named}-*.txt"))) < 2: - await asyncio.sleep(0.001) - - -@flow -async def run(agents: tuple[AgentBase], task: str) -> None: - await _both("inside") - await agents[0].new().aturn("echo shared") - await _both("opened") -''' - -#: The one somebody starts, which says how deep the whole thing goes. -OUTER = '''"""Starts the recursion, five levels of it.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow, load - - -@flow -async def run(agents: tuple[AgentBase], task: str) -> None: - await load("deeper")(agents, task, {"left": 4, "trail": "-"}) -''' - - -@pytest.fixture(autouse=True) -def flows(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: - """A project holding a flow that calls itself, and a home nothing wrote to.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - where = tmp_path / "project" - (where / ".humanize/flows").mkdir(parents=True) - written(where / ".humanize/flows", "deeper", DEEPER) - written(where / ".humanize/flows", "outer", OUTER) - monkeypatch.chdir(where) - return where - - -def _branches(flows: Path) -> dict[str, dict[str, Any]]: - """What each level of the recursion said about itself, by the branch it was on.""" - said = [ - json.loads(line) - for line in (flows / "branches.jsonl").read_text().splitlines() - if line - ] - return {one["trail"]: one for one in said} - - -def test_a_flow_that_calls_two_of_itself_at_once_runs_every_branch(flows: Path) -> None: - """Five levels, doubling at each of them, which is thirty-one runs of one flow.""" - Runner("outer", [ShellAgent(CONFIG)]).run("go") - - branches = _branches(flows) - # One per node of a five-level binary tree: the root, and two children apiece. - assert len(branches) == 31 - assert "-LLLL" in branches # and it really went all the way down one of them - assert "-RLRL" in branches - - -def test_what_is_running_inside_a_gathered_call_is_the_branch_it_is_on( - flows: Path, -) -> None: - """Not its sibling, which is running beside it and under neither of them.""" - Runner("outer", [ShellAgent(CONFIG)]).run("go") - - branches = _branches(flows) - # The flow somebody started, then one entry per call that had to be made to get here -- - # and never two entries for one level, however many of that level were running at once. - assert branches["-LRLR"]["running"] == ["outer", *["deeper"] * 5] - assert branches["-LRLR"]["deep"] == [0, 1, 2, 3, 4, 5] - assert branches["-"]["running"] == ["outer", "deeper"] - assert branches["-"]["deep"] == [0, 1] - assert running() == () # and nothing is left on any branch when the run is over - - -def test_the_epic_reads_back_as_the_tree_the_run_actually_was(flows: Path) -> None: - """A record apiece, each inside the one that called it, however deep it went.""" - Runner("outer", [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - (top,) = tree(epic) - - assert top.flow == "deeper" - assert len(top.calls) == 2 # the two it gathered, and neither under the other - assert {one.record for one in top.calls} != {top.record} - walked, at = 1, top - while at.calls: - at, walked = at.calls[0], walked + 1 - assert walked == 5 # five levels of it, read back as five - # Every record of the epic is somewhere in that tree, and each of them once. - seen: list[str] = [] - - def walk(calls: tuple[Any, ...]) -> None: - for one in calls: - seen.append(one.record) - walk(one.calls) - - walk(top.calls) - seen.append(top.record) - assert sorted(seen) == sorted(one.name for one in records(epic)[1:]) - assert len(seen) == len(set(seen)) - - -def test_no_record_is_attributed_to_the_wrong_parent(flows: Path) -> None: - """Each record says which one called it, and that is the one that says it called it.""" - Runner("outer", [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - for record in records(epic)[1:]: - began = next(one for one in events(record) if one["event"] == "began") - called = [ - one - for one in events(epic / began["under"]) - if one["event"] == "called" and one["epic"] == record.name - ] - assert ( - len(called) == 1 - ) # said once, by the record that says this one is under it - - -def test_a_session_opened_at_the_bottom_is_written_where_it_was_opened( - flows: Path, -) -> None: - """Sixteen leaves open sixteen sessions, and no two of them land in one record.""" - Runner("outer", [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - opened = sessions(epic) - - assert len(opened) == 16 # a leaf apiece - assert ( - len({one.record for one in opened}) == 16 - ) # each in the record that opened it - assert all(one.flow == "deeper" for one in opened) - - -def test_a_recursion_stopped_partway_unwinds_every_level_of_itself( - flows: Path, -) -> None: - """What a ctrl+c reaches is every branch of it, each handing its agents back.""" - written( - flows / ".humanize/flows", - "cancels", - '"""Goes deep and is taken down while it is down there."""\n\n' - "import asyncio\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - " going = asyncio.gather(\n" - ' load("waits")(agents, task, {"left": 4}),\n' - ' load("waits")(agents, task, {"left": 4}),\n' - " )\n" - " await asyncio.sleep(0.05)\n" - " going.cancel()\n" - " try:\n" - " await going\n" - " except asyncio.CancelledError:\n" - " pass\n", - ) - written( - flows / ".humanize/flows", - "waits", - '"""Calls itself down and then waits to be taken down."""\n\n' - "import asyncio\n\n" - "from pydantic import BaseModel\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "class Config(BaseModel):\n" - ' """How much further."""\n\n' - " left: int = 0\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str, config: Config | None = None)" - " -> None:\n" - " setting = config or Config()\n" - " if setting.left <= 0:\n" - " await asyncio.sleep(30)\n" - " return\n" - ' await load("waits")(agents, task, {"left": setting.left - 1})\n', - ) - agent = ShellAgent(CONFIG) - - Runner("cancels", [agent]).run("go") - - assert running() == () # every level of both branches came off it - assert agent.epic is None # and the agent was handed back as the run found it - (epic,) = epics() - for record in records(epic)[1:]: - ended = [one for one in events(record) if one["event"] == "ended"] - assert [one["how"] for one in ended] == ["failed"] - - -def test_a_chain_of_flows_with_no_bottom_to_it_is_refused(flows: Path) -> None: - """A flow that calls itself and never stops is a flow to correct, and says which one.""" - written( - flows / ".humanize/flows", - "forever", - '"""Calls itself, and nothing stops it."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' load("forever")(agents, task)\n', - ) - - with pytest.raises(NotAFlow, match="flows deep"): - Runner("forever", [ShellAgent(CONFIG)]).run("go") - - assert running() == () - - -def test_a_call_may_say_what_the_flow_it_calls_is_driven_at(flows: Path) -> None: - """Which is a clone at that config rather than the agent set up again.""" - written( - flows / ".humanize/flows", - "says", - '"""Says what it was handed."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' Path("says.txt").write_text(\n' - ' agents[0].config.effort + " " + agents[0].id\n' - " )\n", - ) - agent = ShellAgent(CONFIG) - - load("says")([agent], "go", drives={agent.id: AgentConfig(model="m", effort="max")}) - - said = (flows / "says.txt").read_text() - assert said.startswith("max ") # driven at what the call said - assert not said.endswith( - agent.id - ) # and by a second agent, which is what a clone is - assert agent.config.effort == "high" # the caller's own is untouched - - -def test_a_call_that_says_what_it_drives_names_a_place_the_flow_has( - flows: Path, -) -> None: - """Said where the call was written, rather than caught by the flow it was aimed at.""" - written( - flows / ".humanize/flows", - "named", - '"""One agent, and it says what it calls it."""\n\n' - "from typing import NamedTuple\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "class One(NamedTuple):\n" - ' """The one."""\n\n' - " builder: AgentBase\n\n\n" - "@flow\n" - "def run(agents: One, task: str) -> None:\n" - " pass\n", - ) - config = AgentConfig(model="m", effort="max") - - load("named")([ShellAgent(CONFIG)], "go", drives={"builder": config}) - - with pytest.raises(NotAFlow, match="nothing it drives is called 'reviewer'"): - load("named")([ShellAgent(CONFIG)], "go", drives={"reviewer": config}) - - -def test_two_calls_sharing_one_agent_leave_it_where_they_were_both_called_from( - flows: Path, -) -> None: - """Neither of them may have it: a session it opens is part of the flow they share.""" - written( - flows / ".humanize/flows", - "shares", - '"""Gathers two calls over one agent, and neither gets it."""\n\n' - "import asyncio\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - " await asyncio.gather(\n" - ' load("opens")(agents, task),\n' - ' load("opens")(agents, task),\n' - " )\n", - ) - written(flows / ".humanize/flows", "opens", OPENS) - agent = ShellAgent(CONFIG) - - Runner("shares", [agent]).run("go") - - (epic,) = epics() - opened = sessions(epic) - - assert len(opened) == 2 - # Both in the run's own record, which is the flow that called them: writing one of them - # into a sibling's record would be filing it under a flow that happened to be there. - assert {one.record for one in opened} == {"epic.jsonl"} - assert {one.flow for one in opened} == {"shares"} - - -def test_two_calls_sharing_an_agent_leave_it_in_the_record_of_the_flow_they_share( - flows: Path, -) -> None: - """The flow they were both called from, which may be five flows down and not the run.""" - written( - flows / ".humanize/flows", - "top", - '"""Calls the one that gathers, so that the fork is not the run itself."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' await load("shares")(agents, task)\n', - ) - written( - flows / ".humanize/flows", - "shares", - '"""Gathers two calls over the one agent it was handed."""\n\n' - "import asyncio\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - " await asyncio.gather(\n" - ' load("opens")(agents, task),\n' - ' load("opens")(agents, task),\n' - " )\n", - ) - written(flows / ".humanize/flows", "opens", OPENS) - - Runner("top", [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - opened = sessions(epic) - - assert len(opened) == 2 - # In `shares`, which called them both -- not in the run's own record, which is two flows - # further out than the flow they actually share. - assert {one.flow for one in opened} == {"shares"} - assert len({one.record for one in opened}) == 1 - assert opened[0].record != "epic.jsonl" - - -def test_a_call_driving_agents_of_its_own_is_still_part_of_the_run(flows: Path) -> None: - """A branch with none of the run's agents left must not be a branch written down nowhere.""" - written( - flows / ".humanize/flows", - "aims", - '"""Calls one flow at another effort, straight from the run\'s own flow."""\n\n' - "from dataclasses import replace\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' careful = replace(agents[0].config, effort="max")\n' - ' load("opened")(agents, task, drives={agents[0].id: careful})\n', - ) - written( - flows / ".humanize/flows", - "opened", - '"""Opens a session, so the record has something in it."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " agents[0].new()('echo aimed')\n", - ) - - Runner("aims", [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - (one,) = tree(epic) - - assert one.flow == "opened" # the call was written down at all - (session,) = sessions(epic) - assert session.record == one.record # and what the clone opened is in that record - assert session.flow == "opened" - - -def test_a_flow_called_from_a_thread_with_no_branch_on_it_is_still_written_down( - flows: Path, -) -> None: - """A tool a turn reached for is the flow's own code on somebody else's thread.""" - written( - flows / ".humanize/flows", - "elsewhere", - '"""Calls a flow from a thread of its own, the way a tool callback would."""\n\n' - "from concurrent.futures import ThreadPoolExecutor\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " with ThreadPoolExecutor(max_workers=1) as apart:\n" - ' apart.submit(load("opened"), agents, task).result()\n', - ) - written( - flows / ".humanize/flows", - "opened", - '"""Opens a session, so the record has something in it."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " agents[0].new()('echo apart')\n", - ) - - Runner("elsewhere", [ShellAgent(CONFIG)]).run("go") - - (epic,) = epics() - (one,) = tree(epic) - - assert one.flow == "opened" # written down, rather than lost with the thread - (session,) = sessions(epic) - assert session.record == one.record - - -def test_a_gathered_call_cancelled_before_it_starts_takes_nothing(flows: Path) -> None: - """A coroutine that never ran is a call that never happened, holding nothing.""" - written( - flows / ".humanize/flows", - "drops", - '"""Gathers two calls and takes them down before either got a step."""\n\n' - "import asyncio\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - " going = asyncio.gather(\n" - ' load("naps")(agents, task),\n' - ' load("naps")(agents, task),\n' - " )\n" - " going.cancel()\n" - " try:\n" - " await going\n" - " except asyncio.CancelledError:\n" - " pass\n", - ) - written( - flows / ".humanize/flows", - "naps", - '"""Waits, and is never let to."""\n\n' - "import asyncio\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - " await asyncio.sleep(30)\n", - ) - agent = ShellAgent(CONFIG) - - Runner("drops", [agent]).run("go") - - assert running() == () # nothing left on any branch - assert agent.epic is None # and the agent was handed back - (epic,) = epics() - # A call that never ran is a call that was never written down, rather than one with a - # `began` and no end to it. - assert [one.name for one in records(epic)] == ["epic.jsonl"] - - -def test_a_called_flow_hands_its_agents_back_however_two_calls_end(flows: Path) -> None: - """Back to what they were before the first of them took them, not to what one saw.""" - written( - flows / ".humanize/flows", - "carries", - '"""Gathers two calls that each bring skills of their own."""\n\n' - "import asyncio\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - " await asyncio.gather(\n" - ' load("brings")(agents, task),\n' - ' load("brings")(agents, task),\n' - " )\n", - ) - card = "---\nname: brought\ndescription: does a thing\n---\n\nDo it.\n" - written( - flows / ".humanize/flows", - "brings", - '"""Brings a skill of its own."""\n\n' - "import asyncio\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "async def run(agents: tuple[AgentBase], task: str) -> None:\n" - " await asyncio.sleep(0.02)\n", - {"brought": card}, - ) - agent = ShellAgent(CONFIG) - agent.loads([Loaded("mine", flows)]) - - asyncio.run(_gathered(agent)) - - assert [one.name for one in agent.loaded] == ["mine"] - - -async def _gathered(agent: ShellAgent) -> None: - """Runs the flow that gathers two calls over one agent, from a loop of this test's own.""" - answered = load("carries")([agent], "go") - assert answered is not None - await answered diff --git a/tests/integration/flows/test_reloading_flows.py b/tests/integration/flows/test_reloading_flows.py deleted file mode 100644 index eac7e460..00000000 --- a/tests/integration/flows/test_reloading_flows.py +++ /dev/null @@ -1,481 +0,0 @@ -"""A flow rewritten while a run is going, and read again as it is now. - -A flow is a directory on disk rather than a module somebody imported, and everything that -reads one reads it by running its entry point. So a flow edited between two readings of it -- -by hand, or by an agent the flow is itself driving -- is the flow that runs next, and that is -the whole reason a run can improve the thing it is being run by. - -Which is easy to say and easy to lose: one `sys.modules` entry left behind, one entry point -held onto past the call that found it, and a run would go on driving the flow it read the -first time for the life of the process. Everything a flow says about itself is read the same -way, so each of them is checked here after the file under it has changed -- what it drives, -what it can be set up with, whether it can be picked up, what it says it is, what it brings, -and whether it is a flow at all. -""" - -from __future__ import annotations - -import sys -from typing import TYPE_CHECKING - -import pytest - -from hmz._legacy_flows import NotAFlow, load -from hmz.coganchor.agents import AgentConfig -from hmz.runtime.flowing import about, configures, drives, find, held, loaded, resumes -from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent, written - -if TYPE_CHECKING: - from pathlib import Path - -CONFIG = AgentConfig(model="m", effort="high") - - -def _agent() -> ShellAgent: - """One stand-in agent, since none of these flows takes a turn worth watching.""" - return ShellAgent(CONFIG) - - -def _writes(said: str) -> str: - """A flow of one line: it writes what it was told into a file and stops. - - Args: - said: What it writes, which is what a test reads back to say which reading ran. - - Returns: - The flow, as the source of its `__init__.py`. - """ - return ( - '"""Says which reading of it ran."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - f' Path("said.txt").write_text({said!r})\n' - ) - - -@pytest.fixture(autouse=True) -def project(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: - """A project with flows of its own, and a home nothing has written to.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - where = tmp_path / "project" - (where / ".humanize/flows").mkdir(parents=True) - monkeypatch.chdir(where) - return where - - -def test_a_flow_rewritten_between_two_calls_of_one_handle_is_read_again( - project: Path, -) -> None: - """`load` holds the name rather than the entry point it found under it. - - Which is the whole of the promise: the handle is taken once, and each call of it runs the - file as it stands rather than as it stood when somebody asked for it. - """ - flows = project / ".humanize/flows" - written(flows, "one", _writes("first")) - calling = load("one") - - calling([_agent()], "go") - assert (project / "said.txt").read_text() == "first" - - written(flows, "one", _writes("second")) - calling([_agent()], "go") - - assert (project / "said.txt").read_text() == "second" - - -def test_a_flow_that_grows_a_setting_between_calls_is_set_up_by_the_model_it_has_now( - project: Path, -) -> None: - """A config is read back through the model this reading declared, and not the last one. - - Two readings of one file are two classes, and what survives that is the fields. So a - field added to a flow between two calls of it is a field the second call takes -- and a - setting the flow no longer has is one it refuses, which is the same rule read the other - way. - """ - flows = project / ".humanize/flows" - settable = ( - '"""Takes a setting, and writes down what it was set up with."""\n\n' - "import json\n" - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n" - "from pydantic import BaseModel\n\n\n" - "class Config(BaseModel):\n" - ' model_config = {{"extra": "forbid"}}\n\n' - "{fields}\n\n" - "@flow\n" - "def run(\n" - " agents: tuple[AgentBase], task: str, config: Config | None = None\n" - ") -> None:\n" - ' Path("set_up.json").write_text(\n' - " json.dumps({{}} if config is None else config.model_dump())\n" - " )\n" - ) - written(flows, "settable", settable.format(fields=" rounds: int = 3")) - calling = load("settable") - - calling([_agent()], "go", {"rounds": 9}) - assert (project / "set_up.json").read_text() == '{"rounds": 9}' - - # A field the flow did not have when the handle was taken. - written( - flows, - "settable", - settable.format(fields=" rounds: int = 3\n loud: bool = False"), - ) - calling([_agent()], "go", {"rounds": 9, "loud": True}) - - assert (project / "set_up.json").read_text() == '{"rounds": 9, "loud": true}' - assert configures("local/settable") is not None - - # And rewritten to take nothing at all, the same call is one to correct. - written(flows, "settable", _writes("plain")) - with pytest.raises(NotAFlow, match="takes no config"): - calling([_agent()], "go", {"rounds": 9}) - assert configures("local/settable") is None - - -def test_a_flow_that_changes_how_many_agents_it_drives_is_held_to_the_count_it_has_now( - project: Path, -) -> None: - """The count is read at the call, so a flow that grew a place is short of an agent.""" - flows = project / ".humanize/flows" - written(flows, "pair", _writes("one agent")) - calling = load("pair") - - calling([_agent()], "go") - assert (project / "said.txt").read_text() == "one agent" - - written( - flows, - "pair", - '"""Two agents now."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase, AgentBase], task: str) -> None:\n" - " pass\n", - ) - - with pytest.raises(NotAFlow, match="drives 2 agents, 1 given"): - calling([_agent()], "go") - assert drives("local/pair") == ("", "") - - -def test_a_flow_rewritten_into_something_that_is_no_longer_one_says_so_at_the_call( - project: Path, -) -> None: - """Refused where it is asked for, and again at each call: the file may have moved on. - - A name that was never a flow is refused at `load`, which is the line to correct. One that - was a flow and has stopped being one is refused at the call it stopped being one before, - since that is when it happened. - """ - flows = project / ".humanize/flows" - written(flows, "was", _writes("a flow")) - calling = load("was") - calling([_agent()], "go") - - written( - flows, - "was", - '"""Nothing in here is marked."""\n\n\ndef run() -> None:\n pass\n', - ) - - with pytest.raises(NotAFlow, match="nothing in it is marked"): - calling([_agent()], "go") - - # And back again, without the handle being taken a second time. - written(flows, "was", _writes("a flow again")) - calling([_agent()], "go") - - assert (project / "said.txt").read_text() == "a flow again" - - -def test_a_file_that_will_not_run_holds_no_flows_and_holds_them_again_once_it_does( - project: Path, -) -> None: - """A list of flows is drawn while a file is being edited, and must not end on one.""" - flows = project / ".humanize/flows" - at = written(flows, "broken", _writes("fine")) - assert [one.about for one in held(at)] == ["Says which reading of it ran."] - - written(flows, "broken", "this is not python at all(\n") - assert held(at) == [] - - written(flows, "broken", _writes("fine again")) - assert [one.about for one in held(at)] == ["Says which reading of it ran."] - - -def test_a_flow_that_becomes_resumable_between_two_runs_is_handed_state_at_the_next( - project: Path, -) -> None: - """What can happen next is what the flow says today, and not what a run of it recorded.""" - flows = project / ".humanize/flows" - written(flows, "counts", _writes("no state")) - assert not resumes("local/counts") - - written( - flows, - "counts", - '"""Counts its runs."""\n\n' - "from pathlib import Path\n" - "from typing import Any\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow(resumable=True)\n" - "def run(\n" - " agents: tuple[AgentBase], task: str, state: dict[str, Any]\n" - ") -> None:\n" - ' state["rounds"] = state.get("rounds", 0) + 1\n' - ' Path("rounds.txt").write_text(str(state["rounds"]))\n', - ) - - assert resumes("local/counts") - Runner("local/counts", [_agent()]).run("go") - assert (project / "rounds.txt").read_text() == "1" - Runner("local/counts", [_agent()]).run("go") - - assert (project / "rounds.txt").read_text() == "2" - - -def test_what_a_flow_says_about_itself_is_read_as_it_says_it_now(project: Path) -> None: - """The one line a picker shows is read by running the file, like everything else.""" - flows = project / ".humanize/flows" - written(flows, "says", _writes("first")) - assert about("local/says") == "Says which reading of it ran." - - written( - flows, - "says", - '"""It says something else now."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " pass\n", - ) - - assert about("local/says") == "It says something else now." - - -def test_a_module_beside_a_flow_rewritten_between_two_calls_is_read_again( - project: Path, -) -> None: - """A flow is its directory, so what it imports out of it is read again with it. - - The flow file itself is run rather than imported and so was never cached; the module - beside it is imported, and is the half that would be left in `sys.modules` for the life - of the process. A loop that improves its own prompts improves the file they are in. - """ - flows = project / ".humanize/flows" - at = written( - flows, - "reads", - '"""Writes down what the module beside it says."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n" - "import beside\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " Path('said.txt').write_text(beside.SAID)\n", - ) - (at / "beside.py").write_text('SAID = "first"\n') - calling = load("reads") - - calling([_agent()], "go") - assert (project / "said.txt").read_text() == "first" - - (at / "beside.py").write_text('SAID = "second"\n') - calling([_agent()], "go") - - assert (project / "said.txt").read_text() == "second" - - -def test_a_module_beside_a_flow_is_forgotten_so_the_next_flow_reads_its_own( - project: Path, -) -> None: - """Two flows may each keep a `prompts.py`, and the first read must not own the name.""" - flows = project / ".humanize/flows" - for name, said in (("alpha", "alpha"), ("beta", "beta")): - at = written( - flows, - name, - '"""Writes down what the module beside it says."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n" - "import beside\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " Path('said.txt').write_text(beside.SAID)\n", - ) - (at / "beside.py").write_text(f'SAID = "{said}"\n') - - load("alpha")([_agent()], "go") - assert (project / "said.txt").read_text() == "alpha" - load("beta")([_agent()], "go") - assert (project / "said.txt").read_text() == "beta" - load("alpha")([_agent()], "go") - - assert (project / "said.txt").read_text() == "alpha" - assert "beside" not in sys.modules - - -def test_reading_a_flow_leaves_nothing_of_it_behind(project: Path) -> None: - """Neither the flow nor what it imports, and neither in `sys.modules` nor on the path.""" - flows = project / ".humanize/flows" - at = written( - flows, - "leaves", - '"""Imports the module beside it."""\n\n' - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n" - "import beside\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " pass\n", - ) - (at / "beside.py").write_text("SAID = 1\n") - path = list(sys.path) - - assert loaded(find("local/leaves"))["beside"].SAID == 1 - - assert "beside" not in sys.modules - assert sys.path == path - # And humanize's own is untouched: a flow kept inside this tree would otherwise unload - # the package that is running it. - assert "hmz._legacy_flows" in sys.modules - - -def test_a_flow_rewritten_by_the_run_it_is_driving_is_the_one_that_runs_next( - project: Path, -) -> None: - """Which is what a run that improves its own flow comes to. - - The flow rewrites its own file mid-run and then calls itself by name. The call reads what - is on disk now, so the second reading is the rewritten one -- and the run going on is - still the first reading, which is what the entry point that is running has to be. - """ - flows = project / ".humanize/flows" - written(flows, "target", _writes("as it was")) - written( - flows, - "improves", - '"""Rewrites the flow it is about to call."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow, load\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' calling = load("target")\n' - " calling(agents, task)\n" - ' Path("was.txt").write_text(Path("said.txt").read_text())\n' - ' at = Path(".humanize/flows/target/__init__.py")\n' - " at.write_text(at.read_text().replace('as it was', 'as it became'))\n" - " calling(agents, task)\n", - ) - - Runner("local/improves", [_agent()]).run("go") - - assert (project / "was.txt").read_text() == "as it was" - assert (project / "said.txt").read_text() == "as it became" - - -def test_a_flow_a_run_is_driving_is_read_again_by_the_run_after_it( - project: Path, -) -> None: - """A `Runner` is one run and holds the entry point it started; the next one reads again.""" - flows = project / ".humanize/flows" - written(flows, "twice", _writes("first run")) - - Runner("local/twice", [_agent()]).run("go") - assert (project / "said.txt").read_text() == "first run" - - written(flows, "twice", _writes("second run")) - Runner("local/twice", [_agent()]).run("go") - - assert (project / "said.txt").read_text() == "second run" - - -def test_a_flow_that_writes_itself_a_skill_carries_it_at_the_next_call( - project: Path, -) -> None: - """The skills are worked out afresh at each call, being as much the flow as the file is.""" - flows = project / ".humanize/flows" - written( - flows, - "brings", - '"""Writes down what it is carrying."""\n\n' - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " (agent,) = agents\n" - " Path('carried.txt').write_text(\n" - " ','.join(one.name for one in agent.loaded)\n" - " )\n", - skills={"note-taking": "---\nname: note-taking\n---\n"}, - ) - calling = load("brings") - - calling([_agent()], "go") - assert (project / "carried.txt").read_text() == "note-taking" - - written( - flows, - "brings", - (flows / "brings/__init__.py").read_text(), - skills={ - "note-taking": "---\nname: note-taking\n---\n", - "reviewing": "---\n---\n", - }, - ) - calling([_agent()], "go") - - assert (project / "carried.txt").read_text() == "note-taking,reviewing" - - -def test_a_flow_read_again_within_one_second_is_read_as_it_is_now( - project: Path, -) -> None: - """Nothing here is mtime-keyed, so two readings a moment apart are two readings. - - A flow rewritten by the agent it is driving is rewritten in the middle of a loop, which is - where two edits within a second of each other actually happen. A cache keyed on when the - file changed would run the first of them twice. - """ - flows = project / ".humanize/flows" - calling = None - for said in ("a", "b", "c", "d"): - written(flows, "quick", _writes(said)) - calling = calling or load("quick") - calling([_agent()], "go") - assert (project / "said.txt").read_text() == said - - -def test_a_flow_deleted_and_written_again_is_found_again(project: Path) -> None: - """A handle survives the file going: the name is what it holds, and the name came back.""" - import shutil - - flows = project / ".humanize/flows" - written(flows, "gone", _writes("here")) - calling = load("gone") - calling([_agent()], "go") - - shutil.rmtree(flows / "gone") - with pytest.raises(NotAFlow): - calling([_agent()], "go") - - written(flows, "gone", _writes("back")) - calling([_agent()], "go") - - assert (project / "said.txt").read_text() == "back" diff --git a/tests/integration/flows/test_session_skills.py b/tests/integration/flows/test_session_skills.py deleted file mode 100644 index 87053fb3..00000000 --- a/tests/integration/flows/test_session_skills.py +++ /dev/null @@ -1,246 +0,0 @@ -"""Which of a flow's skills one conversation carries, and changing it while it runs. - -A flow brings the skills it works by and every session its agents open is given them. Which of -them is the session's own answer: an agent is what it was made as, and a conversation is a -thing that gets somewhere -- one that has finished reading the codebase and started writing the -tests wants the skill about writing them and no longer wants the eight about reading it. - -So a session says which it carries, and may say it again: what is put where the backend reads -it is settled as each turn opens, so a session told between two turns is carrying what it was -told about on the turn after. What the flow brought is not changed by any of it -- that is the -flow's, and the same flow is driving every one of these conversations. -""" - -from __future__ import annotations - -import gc -from typing import TYPE_CHECKING - -from hmz.coganchor.agents import AgentConfig -from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent, written - -if TYPE_CHECKING: - from pathlib import Path - - import pytest - -CONFIG = AgentConfig(model="m", effort="high") - - -class ClaudeAgent(ShellAgent): - """A stand-in whose backend reads a project's own skills out of `.claude/skills`.""" - - -def skill(name: str) -> str: - """One `SKILL.md`, as every one of these CLIs reads a skill.""" - return f"---\nname: {name}\ndescription: does {name}\n---\n\n# {name}\n" - - -#: Three skills, so that carrying some of them is a thing that can be told apart from -#: carrying all of them and from carrying none. -BROUGHT = {one: skill(one) for one in ("reading", "writing", "reviewing")} - -#: What each flow below is written into: a header naming what it drives, and a body. -HEAD = '''"""A flow that says what its sessions are carrying.""" - -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents -''' - - -def _flow(tmp_path: Path, body: str) -> None: - """Writes the flow out, with the three skills beside it.""" - written(tmp_path / ".humanize/flows", "mine", HEAD + body, BROUGHT) - - -def _ran(tmp_path: Path, body: str) -> ShellAgent: - """Runs that flow over one stand-in agent, and answers with the agent it drove.""" - _flow(tmp_path, body) - agent = ClaudeAgent(CONFIG) - Runner("mine", [agent]).run("go") - gc.collect() # whatever sessions the flow let go of - return agent - - -def _listed(tmp_path: Path, name: str) -> list[str]: - """What was in the mount directory when the turn that wrote that file ran.""" - return sorted((tmp_path / name).read_text().split()) - - -def test_a_session_nobody_has_said_anything_about_carries_all_of_them( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Which is what every session of every flow has always carried.""" - monkeypatch.chdir(tmp_path) - - agent = _ran( - tmp_path, - " session = agent.new()\n" - " session('ls .claude/skills > listed.txt')\n" - " Path('said.txt').write_text(','.join(session.skills))\n", - ) - - assert _listed(tmp_path, "listed.txt") == ["reading", "reviewing", "writing"] - # In the flow's own order, which is the order they are on disk. - assert (tmp_path / "said.txt").read_text() == "reading,reviewing,writing" - assert [one.name for one in agent.loaded] == ["reading", "reviewing", "writing"] - - -def test_a_session_told_which_to_carry_carries_those_and_no_others( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """The rest are not in the workspace at all: what is not carried is not there to read.""" - monkeypatch.chdir(tmp_path) - - _ran( - tmp_path, - " session = agent.new()\n" - " session.loads(['writing'])\n" - " session('ls .claude/skills > listed.txt')\n", - ) - - assert _listed(tmp_path, "listed.txt") == ["writing"] - - -def test_a_session_told_between_two_turns_carries_it_on_the_turn_after( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Which is the whole of what makes this the session's rather than the agent's. - - The turn already running is not touched: what a turn is working by is not moved - underneath it. The next one opens carrying what it was told about. - """ - monkeypatch.chdir(tmp_path) - - _ran( - tmp_path, - " session = agent.new()\n" - " session.loads(['reading'])\n" - " session('ls .claude/skills > first.txt')\n" - " session.loads(['writing'])\n" - " session('ls .claude/skills > second.txt')\n" - " session.loads(None)\n" - " session('ls .claude/skills > third.txt')\n", - ) - - assert _listed(tmp_path, "first.txt") == ["reading"] - # The one it was carrying goes, and the one it was told about arrives. - assert _listed(tmp_path, "second.txt") == ["writing"] - # And nothing is every one of them again, which is where a session starts. - assert _listed(tmp_path, "third.txt") == ["reading", "reviewing", "writing"] - - -def test_two_conversations_of_one_agent_carry_different_ones_at_once( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """One agent, two conversations, and each is somewhere different in the work. - - They work in directories of their own here, since two sessions in one workspace share the - one directory the backend reads: what is checked is that each carries its own answer. - """ - monkeypatch.chdir(tmp_path) - (tmp_path / "one").mkdir() - (tmp_path / "two").mkdir() - - _ran( - tmp_path, - " first = agent.new(cwd='one')\n" - " first.loads(['reading'])\n" - " second = agent.new(cwd='two')\n" - " second.loads(['writing', 'reviewing'])\n" - " first('ls .claude/skills > ../first.txt')\n" - " second('ls .claude/skills > ../second.txt')\n" - " Path('said.txt').write_text(\n" - " ';'.join([','.join(first.skills), ','.join(second.skills)])\n" - " )\n", - ) - - assert _listed(tmp_path, "first.txt") == ["reading"] - assert _listed(tmp_path, "second.txt") == ["reviewing", "writing"] - assert (tmp_path / "said.txt").read_text() == "reading;reviewing,writing" - - -def test_a_name_the_flow_does_not_bring_is_not_a_skill_to_invent( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """What a session may carry is the flow's to say, so the rest is carried and it is not. - - A fork of a flow that dropped a skill is the case: a session asking for it by name is a - session carrying what there is rather than a turn that will not run. - """ - monkeypatch.chdir(tmp_path) - - _ran( - tmp_path, - " session = agent.new()\n" - " session.loads(['writing', 'nothing-called-this'])\n" - " session('ls .claude/skills > listed.txt')\n" - " Path('said.txt').write_text(','.join(session.skills))\n", - ) - - assert _listed(tmp_path, "listed.txt") == ["writing"] - assert (tmp_path / "said.txt").read_text() == "writing" - - -def test_a_session_carrying_none_of_them_has_nothing_in_the_workspace( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """An empty list is an answer: this conversation works by what the CLI already has.""" - monkeypatch.chdir(tmp_path) - - _ran( - tmp_path, - " session = agent.new()\n" - " session.loads([])\n" - " session('ls .claude 2>/dev/null > listed.txt; true')\n" - " Path('said.txt').write_text(','.join(session.skills))\n", - ) - - assert _listed(tmp_path, "listed.txt") == [] - assert (tmp_path / "said.txt").read_text() == "" - assert not (tmp_path / ".claude").exists() - - -def test_what_a_session_carries_is_gone_when_the_session_is( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Mounted rather than installed, whichever of them it was carrying.""" - monkeypatch.chdir(tmp_path) - - _ran( - tmp_path, - " session = agent.new()\n" - " session.loads(['reading'])\n" - " session('ls .claude/skills > listed.txt')\n" - " session.close()\n", - ) - - assert _listed(tmp_path, "listed.txt") == ["reading"] - assert not (tmp_path / ".claude").exists() - - -def test_a_session_closed_and_spoken_to_again_carries_what_it_was_told( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A stopped flow that carries on is the case, and what it carries is its own answer.""" - monkeypatch.chdir(tmp_path) - - _ran( - tmp_path, - " session = agent.new()\n" - " session.loads(['reviewing'])\n" - " session('ls .claude/skills > first.txt')\n" - " session.close()\n" - " session('ls .claude/skills > second.txt')\n", - ) - - assert _listed(tmp_path, "first.txt") == ["reviewing"] - assert _listed(tmp_path, "second.txt") == ["reviewing"] diff --git a/tests/integration/flows/test_stepping.py b/tests/integration/flows/test_stepping.py deleted file mode 100644 index 1b6bf2f8..00000000 --- a/tests/integration/flows/test_stepping.py +++ /dev/null @@ -1,560 +0,0 @@ -"""Running an atlas: the prophecy walked a node at a time, and picked up where it stopped. - -An atlas's body is never run. What runs is the graph compiling it made, which is what puts a -run in a position to be stopped and started: the answers are written down as they arrive, so -picking a run up is walking the same graph over the same answers until it reaches the node -that has none. -""" - -from __future__ import annotations - -from typing import TYPE_CHECKING - -import pytest - -from hmz._legacy_flows import NotAFlow -from hmz.coganchor.agents import AgentConfig -from hmz.runtime.epic import epics, state -from hmz.runtime.flowing import PROPHECY, configures, kept, resumes, wanted -from hmz.runtime.flowing.prophesying import prophesied -from hmz.runtime.runner import Runner -from hmz.sdk import Hmz -from tests.stubs import ShellAgent, written - -if TYPE_CHECKING: - from pathlib import Path - -CONFIG = AgentConfig(model="m", effort="high") - -#: An atlas of three nodes that says which of them ran, and can be made to stop in the last. -THREE = '''"""Three nodes, and a file that says which of them ran.""" - -from pathlib import Path -from typing import NamedTuple - -from pydantic import BaseModel, Field - -from hmz._legacy_flows import Agent, atlas, logic, mind - - -class Agents(NamedTuple): - """Who it drives.""" - - writer: Agent - - -class Said(BaseModel): - """What flows between them.""" - - model_config = {"extra": "forbid"} - - text: str = Field(description="what the node had to say") - - -def ran(name: str) -> None: - """Writes down that one node ran, since a run is what a test is watching.""" - at = Path("ran.txt") - at.write_text((at.read_text() if at.exists() else "") + name + " ") - - -@mind -def first(agent: Agent, task: str) -> Said: - """One turn, and the only one.""" - ran("first") - agent.new()("true") - return Said(text=task) - - -@logic -def middle(said: Said) -> Said: - """Reads it, and is kept.""" - ran("middle") - return said - - -@logic(rerun=RERUN) -def last(said: Said) -> None: - """Stops the first run of it, and says whether a run picked up runs it again.""" - ran("last") - if not Path("been.txt").exists(): - Path("been.txt").write_text("yes") - raise RuntimeError("stopped") - - -@atlas -def run(agents: Agents, task: str) -> None: - """Runs the three of them.""" - said = first(agents.writer, task) - held = middle(said) - last(held) -''' - -#: One that loops, so that a node visited twice is two answers rather than one overwritten. -ROUNDS = '''"""Writes until the reading of it says three rounds have been.""" - -from pathlib import Path -from typing import NamedTuple - -from pydantic import BaseModel, Field - -from hmz._legacy_flows import Agent, atlas, logic, mind - - -class Agents(NamedTuple): - """Who it drives.""" - - writer: Agent - - -class Draft(BaseModel): - """What the writer produced, and on which round.""" - - model_config = {"extra": "forbid"} - - text: str = Field(description="the draft") - round: int = Field(default=0, description="which round wrote it") - - -class Verdict(BaseModel): - """Whether the loop is over.""" - - model_config = {"extra": "forbid"} - - done: bool = Field(description="whether three rounds have been") - - -@mind -def write(agent: Agent, task: str) -> Draft: - """One round of it.""" - at = Path("rounds.txt") - said = at.read_text() if at.exists() else "" - at.write_text(said + "w") - agent.new()("true") - return Draft(text=task, round=len(said) + 1) - - -@logic -def judge(said: Draft) -> Verdict: - """Three rounds and it is done.""" - return Verdict(done=said.round >= 3) - - -@atlas -def run(agents: Agents, task: str) -> None: - """Rounds until it is done.""" - draft = write(agents.writer, task) - verdict = judge(draft) - while not verdict.done: - draft = write(agents.writer, task) -''' - -#: And one whose node is a whole atlas of its own, reached beside it and by name. -INNER = '''"""The atlas that is reached by name.""" - -from pathlib import Path -from typing import NamedTuple - -from pydantic import BaseModel, Field - -from hmz._legacy_flows import Agent, atlas, mind - - -class Agents(NamedTuple): - """Who it drives.""" - - writer: Agent - - -class Said(BaseModel): - """What flows through.""" - - model_config = {"extra": "forbid"} - - text: str = Field(description="the text") - - -@mind -def deepen(agent: Agent, said: Said) -> Said: - """One turn, inside a node that is a graph.""" - agent.new()("true") - at = Path("ran.txt") - at.write_text((at.read_text() if at.exists() else "") + "deepen ") - return Said(text=said.text + "!") - - -@atlas -def run(agents: Agents, said: Said) -> Said: - """One node, and it is a turn.""" - out = deepen(agents.writer, said) - return out -''' - -OUTER = '''"""The atlas with a supernode in it, twice over.""" - -from pathlib import Path -from typing import NamedTuple - -from pydantic import BaseModel, Field - -from hmz._legacy_flows import Agent, atlas, logic, sub - - -class Agents(NamedTuple): - """Who it drives.""" - - writer: Agent - - -class Said(BaseModel): - """What flows through.""" - - model_config = {"extra": "forbid"} - - text: str = Field(description="the text") - - -deeper = sub("inner") - - -@logic -def start(task: str) -> Said: - """Opens it.""" - Path("ran.txt").write_text("start ") - return Said(text=task) - - -@atlas(name="twice") -def beside(agents: Agents, said: Said) -> Said: - """A supernode of this file's own, holding one of another file's.""" - once = deeper(agents, said) - return once - - -@logic -def finish(said: Said) -> None: - """Writes what came back out.""" - Path("out.txt").write_text(said.text) - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - said = start(task) - held = beside(agents, said) - finish(held) -''' - - -@pytest.fixture(autouse=True) -def project(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: - """A project with atlases of its own, and a home nothing wrote to.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - where = tmp_path / "project" - (where / ".humanize/flows").mkdir(parents=True) - monkeypatch.chdir(where) - return where - - -def _write(project: Path, name: str, source: str) -> Path: - """Writes one atlas into this project's own flows.""" - return written(project / ".humanize/flows", name, source) - - -def _run(name: str, task: str = "go") -> None: - """Runs one atlas with a stand-in agent, the way a command line runs any flow.""" - Runner(name, [ShellAgent(CONFIG)]).run(task) - - -def test_an_atlas_runs_its_prophecy_rather_than_its_body(project: Path) -> None: - """The body is a declaration, and each node in it is called by the walk over the graph.""" - _write(project, "three", THREE.replace("RERUN", "True")) - - with pytest.raises(RuntimeError): - _run("three") - - assert (project / "ran.txt").read_text() == "first middle last " - - -def test_an_atlas_says_it_can_be_picked_up_without_being_asked(project: Path) -> None: - """A graph is a list of nodes with an answer apiece, so a run of one writes itself down.""" - _write(project, "three", THREE.replace("RERUN", "True")) - - assert resumes("three") is True - - -def test_what_a_run_has_done_is_the_answers_it_has(project: Path) -> None: - """One line per visit to a node, which is what picking the run up walks over again.""" - _write(project, "rounds", ROUNDS) - - _run("rounds") - - held = state(epics()[-1], "rounds") - assert (project / "rounds.txt").read_text() == "www" - # A node visited twice is two answers: a round that overwrote the last round's would be - # a run nothing could be picked up in the middle of. - assert sorted(held["done"]) == [ - "judge#1", - "judge#2", - "judge#3", - "write#1", - "write:2#1", - "write:2#2", - ] - assert held["done"]["write:2#2"] == {"text": "go", "round": 3} - - -def test_the_node_a_run_stopped_inside_is_run_again(project: Path) -> None: - """Work cut off partway was not done, which is what a node says by saying nothing.""" - _write(project, "three", THREE.replace("RERUN", "True")) - with pytest.raises(RuntimeError): - _run("three") - (project / "ran.txt").write_text("") - - _run("three") - - # The two above it answered already, and are picked up rather than taken again. - assert (project / "ran.txt").read_text() == "last " - - -def test_a_node_may_say_it_is_stepped_past_instead(project: Path) -> None: - """One that had its effect before anything could interrupt it, and answers with nothing.""" - _write(project, "three", THREE.replace("RERUN", "False")) - with pytest.raises(RuntimeError): - _run("three") - (project / "ran.txt").write_text("") - - _run("three") - - assert (project / "ran.txt").read_text() == "" - - -def test_a_run_is_picked_up_into_the_same_graph_or_not_at_all(project: Path) -> None: - """An atlas rewritten is a different graph whose nodes happen to share their names.""" - at = _write(project, "three", THREE.replace("RERUN", "True")) - with pytest.raises(RuntimeError): - _run("three") - at.joinpath("__init__.py").write_text( - THREE.replace("RERUN", "True").replace( - " held = middle(said)", - " said = first(agents.writer, task)\n held = middle(said)", - ) - ) - (project / "ran.txt").write_text("") - - _run("three") - - assert (project / "ran.txt").read_text() == "first first middle last " - - -def test_a_supernode_is_the_graph_under_it_walked(project: Path) -> None: - """One node from outside, one run from within, written down beneath the node it is.""" - _write(project, "inner", INNER) - _write(project, "outer", OUTER) - - _run("outer", "hello") - - assert (project / "ran.txt").read_text() == "start deepen " - assert (project / "out.txt").read_text() == "hello!" - assert sorted(state(epics()[-1], "outer")["done"]) == [ - "beside#1", - "beside#1/deeper#1", - "beside#1/deeper#1/deepen#1", - "finish#1", - "start#1", - ] - - -def test_a_body_that_does_not_compile_is_refused_where_it_is_named( - project: Path, -) -> None: - """Before the first turn rather than hours in, which is what an atlas is for.""" - _write( - project, - "three", - THREE.replace("RERUN", "True").replace(" last(held)", " last(said.text)"), - ) - - with pytest.raises(NotAFlow, match="does not compile"): - _run("three") - - -def test_the_prophecy_a_flowverse_ships_is_the_one_that_runs(project: Path) -> None: - """A repository that has been through the compiling has an answer worth carrying.""" - at = _write(project, "three", THREE.replace("RERUN", "True")) - held = prophesied(at).prophecy - assert held is not None - at.joinpath(PROPHECY).write_bytes(kept(held)) - # The source now says one node more, and the shipped graph says it does not. - at.joinpath("__init__.py").write_text( - THREE.replace("RERUN", "True").replace( - " last(held)", " last(held)\n last(held)" - ) - ) - - with pytest.raises(RuntimeError): - _run("three") - - assert (project / "ran.txt").read_text() == "first middle last " - - -def test_a_shipped_prophecy_that_cannot_be_read_is_refused(project: Path) -> None: - """What a flowverse shipped is what it meant to run, so nothing else quietly runs.""" - at = _write(project, "three", THREE.replace("RERUN", "True")) - at.joinpath(PROPHECY).write_bytes(b"nothing here is a prophecy") - - with pytest.raises(NotAFlow, match="cannot be read"): - _run("three") - - -def test_one_is_shipped_and_read_back_through_the_sdk(project: Path) -> None: - """Which is the one call anything that compiles an atlas makes.""" - _write(project, "rounds", ROUNDS) - flows = Hmz().flows - - where = flows.foretell("rounds") - - assert where.endswith(PROPHECY) - said = flows.prophecy("rounds") - assert said is not None - assert [one.at for one in said.nodes] == ["write", "judge", "write:2"] - assert flows.check("rounds") == () - - -def test_an_atlas_may_be_one_file(project: Path) -> None: - """A flow that is one function still is one, and so is an atlas that is one graph.""" - (project / ".humanize/flows/lone.py").write_text(ROUNDS) - - _run("lone") - - assert (project / "rounds.txt").read_text() == "www" - - -def test_a_flow_that_is_one_file_has_nowhere_to_ship_a_prophecy(project: Path) -> None: - """What is beside such a flow is the other flows, and none of it came with this one.""" - (project / ".humanize/flows/lone.py").write_text(ROUNDS) - - with pytest.raises(NotAFlow, match="no directory to ship a prophecy in"): - Hmz().flows.foretell("lone") - - -def test_a_supernode_answers_with_what_its_return_named(project: Path) -> None: - """A graph may answer with something it bound three nodes before it ended.""" - inner = INNER.replace( - """ out = deepen(agents.writer, said) - return out""", - """ out = deepen(agents.writer, said) - more = deepen(agents.writer, out) - return out""", - ) - _write(project, "inner", inner) - _write(project, "outer", OUTER) - - _run("outer", "hello") - - # Two turns were taken, and what came back out is the first one's answer. - assert (project / "ran.txt").read_text() == "start deepen deepen " - assert (project / "out.txt").read_text() == "hello!" - - -#: One that reads the config, to be run with and without being set up. -BOUNDED = '''"""Reads what the run was set up with, or the defaults where it was not.""" - -from pathlib import Path -from typing import NamedTuple - -from pydantic import BaseModel, Field - -from hmz._legacy_flows import Agent, atlas, logic, mind - - -class Agents(NamedTuple): - """Who it drives.""" - - writer: Agent - - -class Config(BaseModel): - """What it takes.""" - - model_config = {"extra": "forbid"} - - rounds: int = Field(default=2, description="how many rounds it may take") - - -class Said(BaseModel): - """What flows.""" - - model_config = {"extra": "forbid"} - - text: str = Field(description="the text") - - -@mind -def first(agent: Agent, task: str) -> Said: - """One turn.""" - agent.new()("true") - return Said(text=task) - - -@logic -def bounded(said: Said, rounds: int) -> None: - """Writes down the bound it was handed.""" - Path("rounds.txt").write_text(str(rounds)) - - -@atlas -def run(agents: Agents, task: str, config: Config | None = None) -> None: - """Says it.""" - said = first(agents.writer, task) - bounded(said, config.rounds) -''' - - -def test_a_run_nobody_set_up_is_handed_the_config_defaults(project: Path) -> None: - """The body has no way to make one, so the model stands in for itself.""" - _write(project, "bounded", BOUNDED) - - _run("bounded") - - assert (project / "rounds.txt").read_text() == "2" - - -def test_a_run_that_was_set_up_is_handed_what_it_was_set_up_with(project: Path) -> None: - """And the defaults are a fallback rather than a ceiling.""" - _write(project, "bounded", BOUNDED) - - Runner("bounded", [ShellAgent(CONFIG)], config={"rounds": 9}).run("go") - - assert (project / "rounds.txt").read_text() == "9" - - -def test_an_atlas_that_will_not_compile_is_refused_before_the_run_is_set_up( - project: Path, -) -> None: - """Rather than from inside one that has pulled an image and opened an epic.""" - _write( - project, - "three", - THREE.replace("RERUN", "True").replace( - " last(held)", " for one in [1, 2]:\n last(held)" - ), - ) - - with pytest.raises(NotAFlow, match="does not compile"): - Runner("three", [ShellAgent(CONFIG)]) - - -def test_reading_a_flow_does_not_compile_it(project: Path) -> None: - """A picker asks three questions of every flow, and an atlas must answer all three.""" - _write( - project, - "three", - THREE.replace("RERUN", "True").replace( - " last(held)", " for one in [1, 2]:\n last(held)" - ), - ) - - # It does not compile, and every one of these is answered off the entry point alone. - assert resumes("three") is True - assert configures("three") is None - assert [one.name for one in wanted("three")] == ["writer"] diff --git a/tests/integration/flows/test_where_agents_work.py b/tests/integration/flows/test_where_agents_work.py deleted file mode 100644 index c46666c2..00000000 --- a/tests/integration/flows/test_where_agents_work.py +++ /dev/null @@ -1,345 +0,0 @@ -"""Where a flow's agents may work, which is the flow's to say rather than a setting. - -A flow is written for one shape of work. One whose agents read this project cannot have one of -them reading somebody else's, and one written to run its tests in a container of a particular -image is not one to be pointed at a colleague's laptop instead. So a place says what it is -- -nothing, `Remote`, or an `Isolated` naming an image -- and everything else is refused before the -first turn rather than discovered by a turn that landed somewhere surprising. - -What a flow needs *of* that place is said the same way and checked here too: `Needs(where=...)` -names what the machine has to come to, and it is asked of the machine's settings rather than of -a machine, so that a place which will not do is refused before an image has been pulled. What -only a live handshake can answer is not asked here -- a machine that turns out not to be what -its settings promised fails as it starts, and this file does not bring one up. -""" - -from __future__ import annotations - -from typing import TYPE_CHECKING - -import pytest - -from hmz._legacy_flows import NotAFlow -from hmz.coganchor.agents import AgentConfig, Isolated, Needs, Remote, anchored -from hmz.coganchor.machines import DockerConfig -from hmz.runtime.flowing import wanted -from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent - -if TYPE_CHECKING: - from pathlib import Path - -CONFIG = AgentConfig(model="m", effort="high") - -#: A flow whose three agents say three different things about where they work. -DECLARED = ''' -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Isolated, Remote -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The three: one that may be sent away, one in a container, one that stays.""" - - builder: Annotated[AgentBase, Remote] - tester: Annotated[AgentBase, Isolated("python:3.12")] - reviewer: AgentBase - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: A flow whose one agent may be sent away, and has to be: the work is not this machine's. -ELSEWHERE = ''' -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs, Remote -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, and what where it works has to come to.""" - - builder: Annotated[AgentBase, Remote, Needs(where=("remote",))] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: One whose work has to happen on a Linux machine nobody here has to have configured. -CONTAINED = ''' -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Isolated, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, in a container of the flow's own naming.""" - - tester: Annotated[ - AgentBase, Isolated("python:3.12"), Needs(where=("isolated", "linux")) - ] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: One that needs a machine somebody here brought up, which an anchor onto one is not. -MANAGED = ''' -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs, Remote -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, somewhere this run may take down again.""" - - builder: Annotated[AgentBase, Remote, Needs(where=("managed",))] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: One whose container could never come to what it asks of it, which is a flow to correct. -IMPOSSIBLE = ''' -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Isolated, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, in a container that is asked to be a Mac.""" - - tester: Annotated[AgentBase, Isolated("python:3.12"), Needs(where=("darwin",))] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: One driven by an agent nobody points anywhere, whose work still has to land elsewhere. -ANYWHERE_ELSE = ''' -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs, Remote -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, somewhere that is not this machine.""" - - builder: Annotated[AgentBase, Remote, Needs(where=("remote",))] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: A flow that says nothing about where its one agent works, which is most flows. -PLAIN = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - pass -""" - - -def _flow(tmp_path: Path, source: str) -> Path: - """Writes a flow out and answers with its path.""" - where = tmp_path / "flow.py" - where.write_text(source) - return where - - -def test_a_flow_says_where_each_of_its_agents_may_work(tmp_path: Path) -> None: - places = wanted(_flow(tmp_path, DECLARED)) - - assert [place.name for place in places] == ["builder", "tester", "reviewer"] - assert places[0].where is Remote - assert places[1].where == Isolated("python:3.12") - assert places[2].where is None # which is this machine, and nothing to configure - - -def test_an_agent_the_flow_did_not_send_away_may_not_be_sent_away( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """The change this makes: a machine used to be a setting anybody could reach for.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, PLAIN) - agent = ShellAgent( - AgentConfig(model="m", effort="high", machine=anchored("ssh://elsewhere")) - ) - - with pytest.raises(NotAFlow, match="runs on this machine"): - Runner(flow, [agent]) - - -def test_an_agent_the_flow_says_is_remote_may_be_pointed_at_a_machine( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, DECLARED) - builder = ShellAgent( - AgentConfig(model="m", effort="high", machine=anchored("ssh://elsewhere")) - ) - agents = [builder, ShellAgent(CONFIG), ShellAgent(CONFIG)] - - Runner(flow, agents) # which is the whole assertion: it is not refused - - assert builder.config.machine == anchored("ssh://elsewhere") - - -def test_an_isolated_agent_is_given_the_container_the_flow_named( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Nobody is asked which image: the flow said it, and that is the whole of the setting.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, DECLARED) - tester = ShellAgent(CONFIG) - assert tester.config.machine is None - - Runner(flow, [ShellAgent(CONFIG), tester, ShellAgent(CONFIG)]) - - machine = tester.config.machine - assert isinstance(machine, DockerConfig) - assert machine.image == "python:3.12" - - -def test_an_isolated_agent_may_not_be_pointed_anywhere( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, DECLARED) - tester = ShellAgent( - AgentConfig(model="m", effort="high", machine=anchored("ssh://elsewhere")) - ) - - with pytest.raises(NotAFlow, match="container of this flow's own"): - Runner(flow, [ShellAgent(CONFIG), tester, ShellAgent(CONFIG)]) - - -def test_an_agent_that_has_already_worked_is_not_moved( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A conversation resumes under the settings it opened with, so it cannot be relocated.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, DECLARED) - tester = ShellAgent(CONFIG) - tester.new()("echo already") - - with pytest.raises(NotAFlow, match="has already opened a session"): - Runner(flow, [ShellAgent(CONFIG), tester, ShellAgent(CONFIG)]) - - -def test_what_a_flow_says_is_read_where_the_agents_are_chosen(tmp_path: Path) -> None: - """So that whoever is choosing them can offer the machine only where it may be given.""" - places = wanted(_flow(tmp_path, PLAIN)) - - assert [place.where for place in places] == [None] - - -def test_a_place_that_needs_a_machine_is_refused_an_agent_pointed_nowhere( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """This machine comes to nothing at all, so anything asked of where the work lands fails.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, ELSEWHERE) - - with pytest.raises(NotAFlow, match="comes to remote, which this machine does not"): - Runner(flow, [ShellAgent(CONFIG)]) - - -def test_a_place_that_needs_a_machine_takes_one_that_comes_to_what_it_asked( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """The other half: the settings already say `remote`, and nothing had to be reached.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, ELSEWHERE) - builder = ShellAgent( - AgentConfig(model="m", effort="high", machine=anchored("ssh://elsewhere")) - ) - - Runner(flow, [builder]) # which is the whole assertion: it is not refused - - -def test_a_place_is_refused_a_machine_whose_settings_do_not_promise_what_it_needs( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Nobody here started the machine an anchor names, so nobody here may take it down.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, MANAGED) - builder = ShellAgent( - AgentConfig(model="m", effort="high", machine=anchored("ssh://elsewhere")) - ) - - with pytest.raises( - NotAFlow, match="comes to managed, which the machine it works on does not" - ): - Runner(flow, [builder]) - - -def test_the_container_a_flow_named_comes_to_what_that_flow_needs_of_it( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Settled and then checked, so a place the flow put in a container is checked in one.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, CONTAINED) - tester = ShellAgent(CONFIG) - - Runner(flow, [tester]) - - assert isinstance(tester.config.machine, DockerConfig) - - -def test_what_a_place_needs_of_where_it_works_is_read_where_the_agents_are_chosen( - tmp_path: Path, -) -> None: - """So that whoever is choosing them can be asked for a machine that would do.""" - places = wanted(_flow(tmp_path, ELSEWHERE)) - - assert places[0].needs == Needs(where=("remote",)) - - -def test_a_place_refused_for_where_it_works_is_not_moved_there_first( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A refused call must hand its agents back as it found them, containers included.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, IMPOSSIBLE) - tester = ShellAgent(CONFIG) - - with pytest.raises(NotAFlow, match="comes to darwin"): - Runner(flow, [tester]) - - assert tester.config.machine is None - - -def test_a_run_put_in_a_container_from_outside_is_where_that_run_works( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Nothing is pointed at it until the run starts, so it is named rather than read.""" - monkeypatch.chdir(tmp_path) - flow = _flow(tmp_path, ANYWHERE_ELSE) - - Runner(flow, [ShellAgent(CONFIG)], container="python:3.12") - - # And without one it is refused, which is the half that says the container did the work. - with pytest.raises(NotAFlow, match="comes to remote, which this machine does not"): - Runner(flow, [ShellAgent(CONFIG)]) diff --git a/tests/integration/layering/test_layering.py b/tests/integration/layering/test_layering.py index 2ce28b84..4782f554 100644 --- a/tests/integration/layering/test_layering.py +++ b/tests/integration/layering/test_layering.py @@ -96,29 +96,26 @@ "hmz.runtime.epic", "hmz.runtime.tracing", }, - # Handing a flow the agents it declared, naming them, and running it under an epic. Also - # reads the `hmz exec` line, which the interface starts a flow from too. What the flow - # says it drives is `flows`'s to answer. + # Handing a flow a driver for every role it declared and running it under an epic. Also + # reads the `hmz exec` line, whose parts the interface starts a flow from too. What the + # flow declares, and the errors a refusal is read off, are the flow API's to answer. "hmz.runtime.runner": { "hmz.coganchor", - "hmz._legacy_flows", + "hmz.flows", "hmz.runtime.epic", "hmz.runtime.flowing", + "hmz.runtime.kept", "hmz.runtime.settings", "hmz.runtime.telemetry", }, # Everything humanize does to a flow that the flow itself never names: where flows come - # from and what each is called, what one says it drives, the two readings that refuse one - # before it can cost anything, compiling an atlas and walking the prophecy it came to, and - # fetching the skills a flow named. It names what a flow is written against, because - # reading a flow means reading the mark and the interfaces it declared its agents with; it - # names the agents those flows drive; and a flow that calls another is written into the - # epic of the run that called it. And the new flow API's engine, the seam its drivers - # are written against, and the reading of the flags that name them, all written against - # `hmz.flows` -- beside the old ones until every way in has moved over. + # from and what each is called, the engine that loads and runs them, the seam its drivers + # are written against, the drivers over coding agent CLIs and machines, the reading of the + # flags that name them, and fetching the skills a flow named. It names what a flow is + # written against, which is what the engine hands a flow and checks it against, and the + # agents the harness drivers drive. "hmz.runtime.flowing": { "hmz.coganchor", - "hmz._legacy_flows", "hmz.flows", "hmz.runtime.epic", "hmz.runtime.telemetry", @@ -129,15 +126,9 @@ # Types, bar three calls -- `flow`, `load` and `Outworlder.new` -- which hand what they # are given to the engine in `hmz.runtime.flowing` when they are made, and never at # import: see HANDED_THROUGH. It names nothing else of humanize's, not even `coganchor`: - # a flow is somebody else's repository, and what it imports is a promise. + # a flow is somebody else's repository, and what it imports is a promise. The flows + # humanize ships, in `hmz.flows.builtin`, are flows like any other and are held to it. "hmz.flows": {"hmz.runtime.flowing"}, - # The old flow API, kept whole under a name nothing new will import until every way in - # has moved to `hmz.flows` and it can go: the interfaces its flows drive, the mark that - # makes a function one of them, the marks an atlas declares its graph with, and the - # vocabulary a turn is described in, handed through from `coganchor`. The one thing it - # names in `hmz.runtime.flowing` is what it hands back to a flow that calls another, and - # it names it for a type checker and never at import: see HANDED_THROUGH. - "hmz._legacy_flows": {"hmz.coganchor", "hmz.runtime.flowing"}, # humanize as one object: one workspace and everything that can be done in it, composed # out of the layers beside it. It is the front door of the runtime rather than a layer of # its own -- everything here is one place several callers would otherwise each have @@ -146,7 +137,7 @@ # keeps `hmz exec` from paying for a tracer. "hmz.runtime.doing": { "hmz.coganchor", - "hmz._legacy_flows", + "hmz.flows", "hmz.runtime.epic", "hmz.runtime.exporting", "hmz.runtime.flowing", @@ -167,9 +158,12 @@ # to and what a token costs -- all of which are one layer now, and all of which the # sheets read to draw what they are about. "hmz.coganchor", - "hmz._legacy_flows", - # What a flow says it drives, which the flow picker asks before it offers an agent for - # a place, and where flows come from, which is the list `/flow` walks. + # What a flow is: the roles, capabilities and budget the flow picker offers and asks + # for, and the errors a run is refused with. + "hmz.flows", + # What a flow declares, which the flow picker asks before it offers an agent for a + # role, where flows come from, which is the list `/flow` walks, and whoever is outside + # a run, which the interface is. "hmz.runtime.flowing", # The runs of this directory, which `/epics` lists and picks one up from, and how big # a bundle of one came out. Writing one is asked of the runtime like everything else @@ -212,16 +206,9 @@ #: engine inside the call and hands the call to it; nothing of the runtime is imported at the #: top of any of its modules, not even for a type checker, which is what #: :func:`test_the_flow_api_reaches_the_runtime_from_inside_a_call_and_nowhere_else` holds it -#: to. :mod:`hmz._legacy_flows` offers `load` the way it offers everything it does not hold: by -#: name, out of :data:`hmz._legacy_flows._ELSEWHERE`, fetched the moment a flow asks for it -#: and never before, and names the runtime for a type checker alone, which is what -#: :func:`test_the_flow_facade_names_the_runtime_for_types_and_never_at_import` holds it to. -#: So the arrow at import time still points one way, and these stay pairs rather than +#: to. So the arrow at import time still points one way, and this stays a pair rather than #: becoming a habit. -HANDED_THROUGH = { - ("hmz.flows", "hmz.runtime.flowing"), - ("hmz._legacy_flows", "hmz.runtime.flowing"), -} +HANDED_THROUGH = {("hmz.flows", "hmz.runtime.flowing")} #: What reaching the target half costs besides: the two modules of the command line that route #: to it, and the front door of the package they route through. All are held to the same bar as @@ -334,31 +321,6 @@ def test_no_two_layers_name_each_other() -> None: ), f"these layers name each other: {both}" -def test_the_flow_facade_names_the_runtime_for_types_and_never_at_import() -> None: - """The one pair pointing both ways points one way when anything is actually running. - - What makes :data:`HANDED_THROUGH` a pair rather than a hole in the rule: the façade may - name the layer above it for a type checker, and must not import it. A flow that never - calls another flow must not load the runtime by importing the module it writes its own - `@flow` with, and `hmz exec` reading a line that names no flow must not either. - """ - tree = ast.parse((SRC / "hmz" / "_legacy_flows" / "__init__.py").read_text()) - typed = { - node - for branch in ast.walk(tree) - if isinstance(branch, ast.If) and ast.unparse(branch.test) == "TYPE_CHECKING" - for node in ast.walk(branch) - } - ran = { - ast.unparse(node) - for node in ast.walk(tree) - if isinstance(node, ast.ImportFrom) - and (node.module or "").startswith("hmz.runtime") - and node not in typed - } - assert not ran, f"the flow façade imports the runtime for real: {ran}" - - def test_the_flow_api_reaches_the_runtime_from_inside_a_call_and_nowhere_else() -> None: """`hmz.flows` names the engine in the three calls that run, and names nothing else. @@ -437,8 +399,8 @@ def test_every_module_at_the_top_is_a_layer_the_table_governs() -> None: """One left out is unchecked, and reads from here exactly like one deliberately exempt. A package whose name starts with one underscore is still a package, and is held to the - table like any other: the old flow API waits at `hmz._legacy_flows` until nothing names it. - Only the dunder files -- `__init__`, `__main__` and the interpreter's cache -- are not layers. + table like any other. Only the dunder files -- `__init__`, `__main__` and the interpreter's + cache -- are not layers. """ named = { f"hmz.{path.stem}" diff --git a/tests/integration/machines/test_isolation.py b/tests/integration/machines/test_isolation.py index 02bf2dad..8535d413 100644 --- a/tests/integration/machines/test_isolation.py +++ b/tests/integration/machines/test_isolation.py @@ -111,10 +111,3 @@ def test_a_workspace_that_is_not_there_is_refused(tmp_path: Path) -> None: with pytest.raises(FileNotFoundError): DockerConfig(image=IMAGE, workspace=str(missing)).create().start() assert not missing.exists() - - -def test_the_flow_reaches_the_container_only_while_the_run_is_in_one() -> None: - """A run on this machine has none, and a flow does what it always did.""" - from hmz._legacy_flows import container - - assert container() is None diff --git a/tests/integration/runtime/test_epics.py b/tests/integration/runtime/test_epics.py index 10e71f92..aa5c4f06 100644 --- a/tests/integration/runtime/test_epics.py +++ b/tests/integration/runtime/test_epics.py @@ -1,4 +1,4 @@ -"""One run of one flow, written down: which agents were driven, and what each of them opened. +"""One run of one flow, written down: what each role was given, and what each of them opened. Nothing else knows that a session was part of a run. The backends log them one at a time, each under an id of its own, and say nothing about whose they were, which account took their turns @@ -6,101 +6,210 @@ wrote down what it opened, and a person can only find the logs of one if the run points at them. -Every agent here is the shell-backed stand-in, so a session is a real process and nothing -more: no coding agent CLI, no network, nothing CI has not got. The one thing that needs more --- what an epic says about the account a session's turns were taken as, which is a supervised -turn and so a kernel that will hand over a tracee -- is in +The agents here are the engine's fakes, where only what a run writes down is asked about, and +the stand-in `claude` of :mod:`tests.recording`, where what a session is called and where its +log is are: a real process and nothing more, no coding agent CLI, no network, nothing CI has +not got. The one thing that needs more -- what an epic says about the account a session's turns +were taken as, which is a supervised turn and so a kernel that will hand over a tracee -- is in `tests/system/runtime/test_epics.py`. """ from __future__ import annotations +import asyncio import json -import subprocess from typing import TYPE_CHECKING, Any import pytest -from hmz.coganchor.agents import AgentConfig, Stopped -from hmz.runtime.epic import JOURNAL, called, epics, linked, opened, read, sessions +from hmz.coganchor.agents import AgentConfig +from hmz.flows import CostExceeded +from hmz.runtime.epic import ( + JOURNAL, + called, + epics, + linked, + opened, + read, + sessions, + tree, +) +from hmz.runtime.flowing.fakes import FakeAgentDriver from hmz.runtime.runner import Runner -from tests.recording import ONE, ClaudeAgent +from tests.recording import AGENT, ONE, TASK, logged, standing_in from tests.stubs import ShellAgent, events, written if TYPE_CHECKING: from pathlib import Path -CONFIG = AgentConfig(model="m", effort="high") - -#: A flow that opens one session per agent, each of which names itself as it lands. +#: A flow that opens one session per role, each of which answers once. FLOW = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow + + +class Agents(AgentCollection): + actor: Agent + reviewer: Agent + + +class Envs(EnvCollection): + here: LocalEnv + + +class Params(FlowParams): + rounds: int = 1 + + +@flow(agents=Agents, envs=Envs, params=Params) +async def flow_(task, *, agents, envs, params, ctx): + for role in ("actor", "reviewer"): + session = await agents[role].spawn(env=envs["here"]) + await agents[role].run(task, session=session) +""" + +#: A flow that raises, with nothing opened. +RAISES = """ +from hmz.flows import AgentCollection, EnvCollection, FlowParams, flow -@flow -def run(agents: tuple[AgentBase, AgentBase], task: str) -> None: - for at, agent in enumerate(agents): - agent.new()(f"echo session-{at}") +@flow(agents=AgentCollection, envs=EnvCollection, params=FlowParams) +async def raises(task, *, agents, envs, params, ctx): + raise RuntimeError(task) """ +#: A flow that waits for as long as it is let, which is what a stop is tried on. +WAITS = """ +import asyncio + +from hmz.flows import AgentCollection, EnvCollection, FlowParams, flow + + +@flow(agents=AgentCollection, envs=EnvCollection, params=FlowParams) +async def waits(task, *, agents, envs, params, ctx): + await asyncio.sleep(60) +""" + +#: A loop with no exit of its own, which is what a budget stops. +LOOPS = """ +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow + + +class Agents(AgentCollection): + builder: Agent + + +class Envs(EnvCollection): + here: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def loops(task, *, agents, envs, params, ctx): + session = await agents["builder"].spawn(env=envs["here"]) + while True: + await agents["builder"].run(task, session=session) +""" + +#: Flows calling flows: two branches at once, one of which goes a level deeper. +TREE = """ +import asyncio + +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow + + +class Agents(AgentCollection): + builder: Agent + + +class Envs(EnvCollection): + here: LocalEnv + + +class Deep(FlowParams): + depth: int = 0 + + +@flow(agents=Agents, envs=Envs, params=Deep) +async def branch(task, *, agents, envs, params, ctx): + session = await agents["builder"].spawn(env=envs["here"]) + await agents["builder"].run(task, session=session) + if params.depth: + await branch(task, agents=agents, envs=envs, params=Deep(depth=params.depth - 1)) + + +@flow(agents=Agents, envs=Envs, params=FlowParams, name="flow") +async def tree(task, *, agents, envs, params, ctx): + await asyncio.gather( + branch("left", agents=agents, envs=envs, params=Deep(depth=1)), + branch("right", agents=agents, envs=envs, params=Deep()), + ) +""" + +#: What every run here may spend. +BUDGET = {"cost": 5} + def _lines(epic: Path) -> list[dict[str, Any]]: """Every event of one epic, in the order it was written.""" return events(epic) +def _run(tmp_path: Path, source: str, task: str = "go", **agents: object) -> Runner: + """A flow written into the test's own directory, handed fakes by role.""" + written(tmp_path, "flow", source) + return Runner(tmp_path / "flow", agents=agents, budget=BUDGET) # pyright: ignore[reportArgumentType] + + def test_a_run_is_one_epic_and_says_what_it_opened( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: - """The whole of it: what was run, by whom, at what, and every session that came of it.""" + """The whole of it: what was run, what each role was given, and every session of it.""" monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", FLOW) - agents = [ - ShellAgent(CONFIG, name="actor"), - ShellAgent(CONFIG, name="reviewer"), - ] - Runner(tmp_path / "flow", agents).run("go") + _run( + tmp_path, + FLOW, + actor=FakeAgentDriver(model="m", effort="high"), + reviewer=FakeAgentDriver("codex", model="n", provider="work"), + ).run("go") (epic,) = epics() - began, *held, ended = _lines(epic) + began, *held, usage, ended = _lines(epic) assert began["flow"] == str(tmp_path / "flow") + assert began["ref"] == "flow:flow_" assert began["task"] == "go" assert began["workspace"] == str(tmp_path.resolve()) assert began["agents"] == [ { "agent": "actor", - "backend": "shell", + "backend": "claude", "model": "m", "effort": "high", - "service_tier": "default", - "permission": "", "provider": "", - "goals": True, - "web_search": None, - "person": False, }, { "agent": "reviewer", - "backend": "shell", - "model": "m", - "effort": "high", - "service_tier": "default", - "permission": "", - "provider": "", - "goals": True, - "web_search": None, - "person": False, + "backend": "codex", + "model": "n", + "effort": "auto", + "provider": "work", }, ] - assert [(said["agent"], said["session"]) for said in held] == [ - ("actor", "session-0"), - ("reviewer", "session-1"), + assert began["params"] == {"rounds": 1} + assert began["budget"] == { + "duration": None, + "cost": 5.0, + "output_tokens": None, + "graceful": True, + } + assert [(said["agent"], said["backend"], said["provider"]) for said in held] == [ + ("actor", "claude", "local"), + ("reviewer", "codex", "work"), ] + assert usage["event"] == "usage" + assert usage["output_tokens"] == 2 assert ended == {"event": "ended", "at": ended["at"], "how": "done"} # And what a trace is gathered by: whose each of those sessions was. - assert opened(epic) == {"actor": ["session-0"], "reviewer": ["session-1"]} + assert set(opened(epic)) == {"actor", "reviewer"} def test_a_second_run_is_a_second_epic( @@ -108,53 +217,57 @@ def test_a_second_run_is_a_second_epic( ) -> None: """An epic is a run and not a workspace: running the flow again is another run.""" monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", FLOW) for _ in range(2): - Runner(tmp_path / "flow", [ShellAgent(CONFIG), ShellAgent(CONFIG)]).run("go") + _run(tmp_path, FLOW, actor=FakeAgentDriver(), reviewer=FakeAgentDriver()).run( + "go" + ) assert len(epics()) == 2 -def test_a_run_that_was_interrupted_says_so( +def test_a_run_that_was_stopped_says_so( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: - """Esc ends a flow, and an epic that ended that way is not one that finished.""" + """A stop ends a flow, and an epic that ended that way is not one that finished.""" + from hmz.runtime.doing.running import Run + + monkeypatch.chdir(tmp_path) + run = Run(_run(tmp_path, WAITS), "go") + + run.start() + while run.epic is None: + run.wait(0.05) + run.stop() + + assert run.wait(10) + assert isinstance(run.raised, asyncio.CancelledError) + (epic,) = epics() + assert _lines(epic)[-1]["how"] == "stopped" + + +def test_a_run_its_budget_stopped_says_so( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """A loop with no exit of its own ends when its budget is spent, which is not a failure.""" monkeypatch.chdir(tmp_path) - written( - tmp_path, - "flow", - "from hmz.coganchor.agents import AgentBase, Stopped\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' raise Stopped("stopped")\n', - ) - with pytest.raises(Stopped): - Runner(tmp_path / "flow", [ShellAgent(CONFIG)]).run("go") + with pytest.raises(CostExceeded): + _run(tmp_path, LOOPS, builder=FakeAgentDriver(cost=2.0)).run("go") (epic,) = epics() assert _lines(epic)[-1]["how"] == "stopped" + assert _lines(epic)[-2]["cost"] == 6.0 def test_a_run_that_failed_says_so( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: - """A turn that failed takes the flow with it, and the epic says how it went.""" + """What a flow raises takes the run with it, and the epic says how it went.""" monkeypatch.chdir(tmp_path) - written( - tmp_path, - "flow", - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' agents[0].new()("exit 3")\n', - ) - with pytest.raises(subprocess.CalledProcessError): - Runner(tmp_path / "flow", [ShellAgent(CONFIG)]).run("go") + with pytest.raises(RuntimeError, match="nope"): + _run(tmp_path, RAISES).run("nope") (epic,) = epics() assert _lines(epic)[-1]["how"] == "failed" @@ -167,9 +280,8 @@ def test_the_epics_of_one_workspace_are_not_another_workspace_s( here, there = tmp_path / "here", tmp_path / "there" for where in (here, there): where.mkdir() - written(where, "flow", FLOW) monkeypatch.chdir(here) - Runner(here / "flow", [ShellAgent(CONFIG), ShellAgent(CONFIG)]).run("go") + _run(here, FLOW, actor=FakeAgentDriver(), reviewer=FakeAgentDriver()).run("go") assert len(epics(here)) == 1 assert epics(there) == [] @@ -177,7 +289,7 @@ def test_the_epics_of_one_workspace_are_not_another_workspace_s( def test_an_agent_driven_by_hand_is_not_a_run_of_anything(tmp_path: Path) -> None: """A session opened outside a flow belongs to no epic, and writes to none.""" - agent = ShellAgent(CONFIG) + agent = ShellAgent(AgentConfig(model="m", effort="high")) agent.new()("echo alone") @@ -189,10 +301,11 @@ def test_a_session_is_named_for_whose_it_is_what_ran_it_and_which_account( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """The id alone says none of that, and a directory of ids is one nobody can read.""" + standing_in(tmp_path, monkeypatch) monkeypatch.chdir(tmp_path) written(tmp_path, "flow", ONE) - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") + Runner(tmp_path / "flow", agents={"builder": AGENT}, budget=BUDGET).run(TASK) (epic,) = epics() (one,) = sessions(epic) @@ -202,33 +315,30 @@ def test_a_session_is_named_for_whose_it_is_what_ran_it_and_which_account( "builder", "claude", "local", - "the-session", - "builder-claude@local-the-session", + one.ident, + f"builder-claude@local-{one.ident}", one.at, str(tmp_path / "flow"), "", # forked from nothing, which is what a session nobody branched is JOURNAL, ) - assert one.name == called("builder", "claude", "", "the-session") + assert one.name == called("builder", "claude", "", one.ident) def test_the_logs_of_a_session_are_linked_into_the_epic_that_opened_it( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """A link rather than a copy: humanize reads and writes the log where the backend keeps it.""" + config = standing_in(tmp_path, monkeypatch) monkeypatch.chdir(tmp_path) written(tmp_path, "flow", ONE) - where = tmp_path / "claude-home" - monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(where)) - log = where / "projects" / "-tmp-project" / "the-session.jsonl" - log.parent.mkdir(parents=True) - log.write_text('{"type":"user"}\n') - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") + Runner(tmp_path / "flow", agents={"builder": AGENT}, budget=BUDGET).run(TASK) (epic,) = epics() (one,) = sessions(epic) - link = epic / "sessions" / one.name / "the-session.jsonl" + log = logged(config, tmp_path.resolve(), one.ident) + link = epic / "sessions" / one.name / log.name assert link.is_symlink() assert link.resolve() == log.resolve() assert linked(epic) == {one.name: [str(log)]} @@ -238,39 +348,33 @@ def test_a_log_written_after_the_last_turn_is_linked_when_the_run_ends( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """A sub-agent's transcript is written whenever that sub-agent ran, which is later.""" + config = standing_in(tmp_path, monkeypatch) monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - where = tmp_path / "claude-home" - monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(where)) - under = where / "projects" / "-tmp-project" - under.mkdir(parents=True) - (under / "the-session.jsonl").write_text("{}\n") - late = under / "the-session" / "subagents" / "deep" / "explore.jsonl" - late.parent.mkdir(parents=True) + monkeypatch.setenv("LOGS_UNDER", str(config / "projects")) # Written after the session was opened, which is when a sub-agent's transcript is # written: the flow stands in for the backend finishing what it was writing. - monkeypatch.setenv("LATE_LOG", str(late)) written( tmp_path, "flow", - "import os\n" - "from pathlib import Path\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' agents[0].new()("echo the-session")\n' - ' Path(os.environ["LATE_LOG"]).write_text("{}\\n")\n', + ONE.replace( + ' return await agents["builder"].run(task, session=session)\n', + ' said = await agents["builder"].run(task, session=session)\n' + " import os, pathlib\n" + ' (log,) = pathlib.Path(os.environ["LOGS_UNDER"]).glob("*/*.jsonl")\n' + ' late = log.with_suffix("") / "subagents" / "deep" / "explore.jsonl"\n' + " late.parent.mkdir(parents=True)\n" + ' late.write_text("{}\\n")\n' + " return said\n", + ), ) - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") + Runner(tmp_path / "flow", agents={"builder": AGENT}, budget=BUDGET).run(TASK) (epic,) = epics() (one,) = sessions(epic) - assert sorted(p.name for p in (epic / "sessions" / one.name).iterdir()) == [ - "explore.jsonl", - "the-session.jsonl", - ] + assert sorted(p.name for p in (epic / "sessions" / one.name).iterdir()) == sorted( + ["explore.jsonl", f"{one.ident}.jsonl"] + ) def test_a_epic_reads_back_as_what_was_run_and_how_it_went( @@ -278,69 +382,52 @@ def test_a_epic_reads_back_as_what_was_run_and_how_it_went( ) -> None: """Which is what a listing of them shows, and what one of them can be picked up from.""" monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") + _run(tmp_path, FLOW, actor=FakeAgentDriver(), reviewer=FakeAgentDriver()).run("go") (epic,) = epics() ran = read(epic) assert ran is not None - assert (ran.flow, ran.task, ran.how) == (str(tmp_path / "flow"), "go", "done") + assert (ran.flow, ran.ref, ran.task, ran.how) == ( + str(tmp_path / "flow"), + "flow:flow_", + "go", + "done", + ) assert ran.workspace == str(tmp_path.resolve()) assert ran.name == epic.name assert not ran.resumable - assert [one.agent for one in ran.agents] == ["builder"] - assert [one.ident for one in ran.sessions] == ["the-session"] + assert [one.agent for one in ran.agents] == ["actor", "reviewer"] + assert [one.agent for one in ran.sessions] == ["actor", "reviewer"] + assert ran.params == {"rounds": 1} + assert ran.budget is not None + assert ran.budget["cost"] == 5.0 -def test_what_an_agent_was_allowed_reads_as_what_that_agent_actually_did( - tmp_path: Path, +def test_flows_calling_flows_read_back_as_the_tree_they_ran_in( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: - """Two silences in one field, and a report that read them as one would invent a rung.""" - at = tmp_path / "epic" - at.mkdir() - lines: tuple[dict[str, Any], ...] = ( - { - "event": "began", - "at": "1", - "flow": "f", - "task": "go", - "agents": [ - # Written before a run wrote this down at all, which is every run there was - # when every agent was allowed everything. - {"agent": "older", "backend": "shell", "model": "m", "effort": "high"}, - # And written by a run that said nothing to the CLI about what its agent - # may do, which is a rung nobody chose rather than a record gone missing. - { - "agent": "quiet", - "backend": "shell", - "model": "m", - "effort": "high", - "permission": "", - }, - { - "agent": "narrowed", - "backend": "shell", - "model": "m", - "effort": "high", - "permission": "read-only", - }, - ], - }, - {"event": "ended", "at": "2", "how": "done"}, - ) - (at / JOURNAL).write_text( - "\n".join(json.dumps(one) for one in lines), encoding="utf-8" - ) + """Each call a record of its own under the one that made it, each session in its own. - ran = read(at) + Two branches at once, one of them a level deeper: read as a list it would be three things + one run did, and nothing would say which of them ran under which. + """ + monkeypatch.chdir(tmp_path) - assert ran is not None - assert [(one.agent, one.permission) for one in ran.agents] == [ - ("older", "bypass"), - ("quiet", ""), - ("narrowed", "read-only"), + _run(tmp_path, TREE, builder=FakeAgentDriver()).run("go") + + (epic,) = epics() + calls = tree(epic) + assert sorted((one.flow, len(one.calls), one.how) for one in calls) == [ + ("flow:branch", 0, "done"), + ("flow:branch", 1, "done"), ] + (deeper,) = [one for one in calls if one.calls] + assert [(one.flow, one.how) for one in deeper.calls] == [("flow:branch", "done")] + # Every session in the record of the call that opened it, and none in the run's own. + records = {one.record for one in sessions(epic)} + assert JOURNAL not in records + assert len(records) == 3 def test_a_run_written_before_calls_had_records_still_reads_as_what_it_called( diff --git a/tests/integration/runtime/test_export.py b/tests/integration/runtime/test_export.py index 677db79d..4be54d26 100644 --- a/tests/integration/runtime/test_export.py +++ b/tests/integration/runtime/test_export.py @@ -11,22 +11,23 @@ under that -- held to what a bundle holds and what it must never carry rather than to how anything asked for one. -Every run here is driven by the shell-backed stand-in, so the whole file is a real process, a -temporary home and a tarball: nothing CI has not got. The one check that needs more -- that a -run taken as a named account carries none of that account's key, which means a supervised turn -and so a kernel that will hand over a tracee -- is in `tests/system/runtime/test_export.py`. +Every run here is driven by a stand-in CLI -- `claude`, which keeps its conversations where +Claude Code does, and `opencode`, which keeps them to itself -- so the whole file is a real +process, a temporary home and a tarball: nothing CI has not got. The one check that needs +more -- that a run taken as a named account carries none of that account's key, which means a +supervised turn and so a kernel that will hand over a tracee -- is in +`tests/system/runtime/test_export.py`. """ from __future__ import annotations import json import tarfile -from typing import TYPE_CHECKING +from typing import TYPE_CHECKING, Any import pytest from hmz.coganchor import providers -from hmz.coganchor.agents import AgentConfig from hmz.runtime.epic import epics, sessions from hmz.runtime.exporting import ( MANIFEST, @@ -38,134 +39,140 @@ sized, ) from hmz.runtime.runner import Runner -from tests.recording import ONE, ClaudeAgent, claude_home, held, manifest -from tests.stubs import ShellAgent, written +from tests.flows import standins +from tests.recording import AGENT, ONE, TASK, held, manifest, standing_in +from tests.recording import logged as kept +from tests.stubs import written if TYPE_CHECKING: from pathlib import Path -CONFIG = AgentConfig(model="m", effort="high") - -#: A flow that opens one session per agent, each naming itself as it lands. +#: A flow that opens one session per agent. FLOW = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow -@flow -def run(agents: tuple[AgentBase, AgentBase], task: str) -> None: - for at, agent in enumerate(agents): - agent.new()(f"echo session-{at}") -""" +class Agents(AgentCollection): + actor: Agent + reviewer: Agent -#: A flow that calls another, so that a bundle has a record beside the run's own to carry. -CALLS = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow, load +class Envs(EnvCollection): + here: LocalEnv -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - load("under")(agents, task) + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def flow_(task, *, agents, envs, params, ctx): + for role in ("actor", "reviewer"): + session = await agents[role].spawn(env=envs["here"]) + await agents[role].run(task, session=session) """ -#: The one it calls, which opens a session of its own -- one run, two records. -UNDER = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +#: A flow that calls another, so that a bundle has a record beside the run's own to carry; +#: and calls it twice, so that two calls of one flow are two records and the manifest has to +#: say which of them a session belongs to. +CALLS = """ +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow, load + +class Agents(AgentCollection): + builder: Agent -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - agents[0].new()("echo the-session") -""" -#: One that calls the same flow twice, so that two calls of one flow are two records and the -#: manifest has to say which of them a session belongs to. -TWICE = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow, load +class Envs(EnvCollection): + here: LocalEnv -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - load("each")(agents, "once") - load("each")(agents, "again") +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def flow_(task, *, agents, envs, params, ctx): + for said in TASKS: + await load("under")(said, agents=agents, envs=envs, params=FlowParams()) """ -#: The one it calls twice, whose session is named after the call so the two are two. -EACH = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +#: The one it calls, which opens a session of its own -- one run, more than one record. +UNDER = ONE.replace("async def one(", "async def under(") +#: A flow that keeps something, so that its journal is there to be carried. +KEEPS = """ +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - agents[0].new()(f"echo {task}") -""" -#: A flow that leaves something behind, so that `state.json` is there to be carried. -KEEPS = """ -from typing import Any +class Agents(AgentCollection): + builder: Agent -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +class Envs(EnvCollection): + here: LocalEnv -@flow(resumable=True) -def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None: - state["rounds"] = 3 - agents[0].new()("echo the-session") + +@flow(agents=Agents, envs=Envs, params=FlowParams, resumable=True) +async def flow_(task, *, agents, envs, params, ctx): + ctx.state["rounds"] = 3 + session = await agents["builder"].spawn(env=envs["here"]) + return await agents["builder"].run(task, session=session) """ -class OpencodeAgent(ShellAgent): - """A stand-in for one that keeps its sessions to itself and logs nothing.""" +def _ran( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + source: str = ONE, + task: str = TASK, + **agents: str, +) -> Path: + """Runs one flow, written into the test's own directory, on the stand-ins; its epic.""" + standing_in(tmp_path, monkeypatch) + standins.install(tmp_path / "bin", "opencode", standins.OPENCODE) + monkeypatch.chdir(tmp_path) + written(tmp_path, "flow", source) + Runner( + tmp_path / "flow", agents=agents or {"builder": AGENT}, budget={"cost": 5} + ).run(task) + (epic,) = epics() + return epic + + +def _log(tmp_path: Path, epic: Path) -> Path: + """Where the stand-in kept the one conversation of a run.""" + (one,) = sessions(epic) + return kept(tmp_path / "claude-home", tmp_path.resolve(), one.ident) def test_a_bundle_holds_every_record_the_run_wrote( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """A flow that called another is two records, and one run is both of them.""" - monkeypatch.chdir(tmp_path) written(tmp_path / ".humanize" / "flows", "under", UNDER) - written(tmp_path, "flow", CALLS) - claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch, CALLS.replace("TASKS", f"[{TASK!r}]")) inside = held(bundle(epic, tmp_path / "out.tar.gz")[0]) assert "epic.jsonl" in inside - assert [one for one in inside if one.startswith("epic.under_")], inside + assert [one for one in inside if one.startswith("epic.under-under_")], inside # And the manifest lists what went in, the run's own record first. said = json.loads(inside[MANIFEST]) assert said["held"][0] == "epic.jsonl" assert said["held"][-1] == MANIFEST # And the run's own record reads back as the run: every line it wrote, not a summary. - assert ( - '"event": "began"' in inside["epic.jsonl"] - or '"event":"began"' in (inside["epic.jsonl"]) - ) + assert '"event": "began"' in inside["epic.jsonl"] def test_the_manifest_says_the_call_tree_and_whose_each_session_was( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """A flow called twice is two records and two conversations, and the flow name says one.""" - monkeypatch.chdir(tmp_path) - written(tmp_path / ".humanize" / "flows", "each", EACH) - written(tmp_path, "flow", TWICE) - claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + written(tmp_path / ".humanize" / "flows", "under", UNDER) + epic = _ran( + tmp_path, + monkeypatch, + CALLS.replace( + "TASKS", + '["Reply with the single word: once", "Reply with the single word: again"]', + ), + ) said = manifest(bundle(epic, tmp_path / "out.tar.gz")[0]) # Two calls of one flow, each in a record of its own, each saying which called it. - assert [one["task"] for one in said["called"]] == ["once", "again"] or [ - one["task"] for one in said["called"] - ] == ["again", "once"] + assert [one["flow"] for one in said["called"]] == ["under:under", "under:under"] assert {one["under"] for one in said["called"]} == {"epic.jsonl"} assert len({one["record"] for one in said["called"]}) == 2 # And each session says which of the two it was opened in, not only which flow. @@ -181,32 +188,24 @@ def test_a_session_log_comes_as_its_contents_and_not_as_a_link( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """The whole point: a link into somebody's home is worth nothing on another machine.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - claude_home(tmp_path, monkeypatch, '{"type":"user","text":"hello"}') - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch, task="Reply with the single word: hello") (one,) = sessions(epic) + name = f"{one.ident}.jsonl" at = bundle(epic, tmp_path / "out.tar.gz")[0] with tarfile.open(at) as opened: - member = opened.getmember(f"{epic.name}/sessions/{one.name}/the-session.jsonl") + member = opened.getmember(f"{epic.name}/sessions/{one.name}/{name}") assert not member.issym() assert not member.islnk() assert member.isfile() - assert '"text":"hello"' in held(at)[f"sessions/{one.name}/the-session.jsonl"] + assert "single word: hello" in held(at)[f"sessions/{one.name}/{name}"] def test_a_backend_that_logs_nothing_says_so_rather_than_carrying_nothing( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """A CLI that keeps its sessions in a database logs none, and an absence reads as loss.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - - Runner(tmp_path / "flow", [OpencodeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch, builder="opencode/opencode/big-pickle:high") said = manifest(bundle(epic, tmp_path / "out.tar.gz")[0]) (one,) = said["sessions"] @@ -224,16 +223,8 @@ def test_what_the_run_says_it_ran_is_not_struck_out( A bundle with the model taken out of it because some other account had that name in a variable is a bundle saying nothing about what actually ran. """ - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) providers.add("kimi", "elsewhere", "env", {"KIMI_MODEL_NAME": "fixture-model"}) - claude_home(tmp_path, monkeypatch) - agent = ClaudeAgent( - AgentConfig(model="fixture-model", effort="high"), name="builder" - ) - - Runner(tmp_path / "flow", [agent]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch, builder="claude/fixture-model:high") said = manifest(bundle(epic, tmp_path / "out.tar.gz")[0]) assert said["agents"][0]["model"] == "fixture-model" @@ -244,15 +235,13 @@ def test_the_manifest_is_scrubbed_like_everything_else( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """It holds the task, which is a line somebody typed and may hold anything.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) providers.add("claude", "work", "key", {"ANTHROPIC_AUTH_TOKEN": "hunter2-hunter2"}) - claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run( - "push to https://bob:ghp_abcdefghijklmnopqrst@example.com with hunter2-hunter2" + epic = _ran( + tmp_path, + monkeypatch, + task="push to https://bob:ghp_abcdefghijklmnopqrst@example.com with " + "hunter2-hunter2", ) - (epic,) = epics() at, handed = bundle(epic, tmp_path / "out.tar.gz") said = held(at)[MANIFEST] @@ -261,17 +250,15 @@ def test_the_manifest_is_scrubbed_like_everything_else( assert "bob" not in said # And what comes back is what was written, not what was about to be. assert "hunter2" not in json.dumps(handed) + # Nor anywhere else in it: the session's own log said the task too. + assert not any("hunter2" in one for one in held(at).values()) def test_where_a_bundle_lands_beside_two_of_them_at_once( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """Two exports of one run must not be two gzip streams into one file.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - claude_home(tmp_path, monkeypatch) - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch) at = bundle(epic)[0] assert bundle(epic)[0] == at @@ -287,11 +274,7 @@ def test_a_directory_to_fill_that_is_not_there_yet_is_still_a_directory( Answering it with the file would have the next run write over the last. """ - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - claude_home(tmp_path, monkeypatch) - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch) at = bundle(epic, f"{tmp_path / 'bundles'}/")[0] @@ -303,54 +286,46 @@ def test_the_manifest_says_which_run_on_what_and_by_whom( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """Everything a reader needs to know what they are looking at, and nothing secret.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", FLOW) - claude_home(tmp_path, monkeypatch) - - Runner( - tmp_path / "flow", - [ClaudeAgent(CONFIG, name="actor"), ClaudeAgent(CONFIG, name="reviewer")], - ).run("fix the thing") - (epic,) = epics() + epic = _ran( + tmp_path, + monkeypatch, + FLOW, + task="Reply with the single word: fixed", + actor=AGENT, + reviewer=AGENT, + ) said = manifest(bundle(epic, tmp_path / "out.tar.gz")[0]) assert said["epic"] == epic.name - assert said["run"]["task"] == "fix the thing" + assert said["run"]["task"] == "Reply with the single word: fixed" assert said["run"]["how"] == "done" assert said["workspace"]["at"] == str(tmp_path.resolve()) assert [one["agent"] for one in said["agents"]] == ["actor", "reviewer"] - assert said["agents"][0]["runs"] == "claude/m:high" + assert said["agents"][0]["runs"] == "claude/claude-haiku-4-5:low" assert "claude" in said["backends"] assert MANIFEST in said["held"] assert said["humanize"] assert said["redacted"] -def test_what_a_resumable_flow_left_behind_is_in_it( +def test_what_a_resumable_flow_kept_is_in_it( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: - """A run picked up again is picked up from that file, so a report of one needs it.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", KEEPS) - claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + """A run picked up again is picked up from its journal, so a report of one needs it.""" + epic = _ran(tmp_path, monkeypatch, KEEPS) inside = held(bundle(epic, tmp_path / "out.tar.gz")[0]) - assert json.loads(inside["state.json"])[str(tmp_path / "flow")] == {"rounds": 3} + kept_: list[dict[str, Any]] = [ + json.loads(line) for line in inside["resume.jsonl"].splitlines() + ] + assert {"t": "set", "id": 1, "key": "rounds", "value": 3} in kept_ def test_a_trace_gathered_of_the_run_goes_with_it( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """A trace belongs with the run, and so belongs in the bundle of that run.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch) (epic / "traces").mkdir() (epic / "traces" / "a.trace.json").write_text('{"traceEvents": []}', "utf-8") @@ -362,12 +337,7 @@ def test_the_transcript_goes_in_as_it_was_written( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """What the interface hands over is the text, not the rows. There is none from a line.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch) with_screen = held( bundle(epic, tmp_path / "a.tar.gz", transcript="a long line\n")[0] @@ -380,12 +350,7 @@ def test_a_bundle_is_readable_by_whoever_made_it_and_nobody_else( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """What is in it is their prompts and their agents' output.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch) assert bundle(epic)[0].stat().st_mode & 0o777 == 0o600 @@ -394,12 +359,7 @@ def test_a_bundle_carries_nothing_about_whoever_made_it( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """A tar records its writer by default, and a login name is not a thing to hand over.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch) with tarfile.open(bundle(epic, tmp_path / "out.tar.gz")[0]) as opened: assert {one.uname for one in opened.getmembers()} == {""} @@ -408,12 +368,7 @@ def test_a_bundle_carries_nothing_about_whoever_made_it( def test_where_a_bundle_lands(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: """A file outright, a directory to fill, or `.humanize/` here -- it is a thing to send.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch) # Where somebody is standing rather than in humanize's own home the way a trace of a # run goes, and named whole: it is a thing to attach to something. @@ -434,13 +389,8 @@ def test_a_link_whose_log_has_gone_is_left_out_rather_than_carried_empty( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """A log rolls over, a home is thrown away: a name with nothing behind it is not a log.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - log = claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() - log.unlink() + epic = _ran(tmp_path, monkeypatch) + _log(tmp_path, epic).unlink() said = manifest(bundle(epic, tmp_path / "out.tar.gz")[0]) assert said["sessions"][0]["logs"] == [] @@ -451,15 +401,11 @@ def test_what_each_session_was_logged_to_is_read_through_the_links( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: """The files themselves, since what a bundle carries is what is behind each link.""" - monkeypatch.chdir(tmp_path) - written(tmp_path, "flow", ONE) - log = claude_home(tmp_path, monkeypatch) - - Runner(tmp_path / "flow", [ClaudeAgent(CONFIG, name="builder")]).run("go") - (epic,) = epics() + epic = _ran(tmp_path, monkeypatch) (one,) = sessions(epic) + log = _log(tmp_path, epic) - assert logged(epic) == {one.name: {"the-session.jsonl": log.resolve()}} + assert logged(epic) == {one.name: {log.name: log.resolve()}} @pytest.mark.parametrize( diff --git a/tests/integration/runtime/test_resuming.py b/tests/integration/runtime/test_resuming.py index 6e78e22d..de4773b9 100644 --- a/tests/integration/runtime/test_resuming.py +++ b/tests/integration/runtime/test_resuming.py @@ -1,323 +1,184 @@ -"""A flow that says it can be picked up where the last run of it left off. +"""A flow that says it can be picked up where a run of it left off. A loop meant to run for a week is a loop that will be stopped and started: a machine goes down, somebody presses esc, a turn takes the process with it. What such a flow needs is not a second copy of the transcript -- the backends keep that -- but the handful of things it is itself keeping track of: which round it is on, which files it has been through, what it has -decided so far. So a flow says it can be picked up, and is handed a dict: what it wrote there -last time, kept in the run's own epic and saved as it writes. +decided so far. So a flow says it is `resumable`, keeps those in `ctx.state`, and the engine +writes them into a journal inside the run's own epic as it writes them. + +`--resume` -- `resume=True` here -- picks up the newest run of that flow in this workspace that +got as far as writing anything down; without it every run starts from the top. The run that +picks one up is a run of its own, in an epic of its own, handed a copy of the journal it picks +up and saying which run that came from. """ from __future__ import annotations -import json from typing import TYPE_CHECKING import pytest -from hmz._legacy_flows import NotAFlow -from hmz.coganchor.agents import AgentConfig, Stopped -from hmz.runtime.epic import STATE, epics, read, resumed, state -from hmz.runtime.flowing import resumes -from hmz.runtime.runner import Runner -from tests.stubs import ShellAgent, written +from hmz.runtime.epic import RESUME, epics, picks_up, read, resumed, state +from hmz.runtime.runner import Refused, Runner +from tests.stubs import written if TYPE_CHECKING: from pathlib import Path -CONFIG = AgentConfig(model="m", effort="high") - -#: A flow that counts the runs of it, which is the smallest thing a state is for. +#: A flow that counts the runs of it, and fails on the run it is told to. COUNTS = '''"""Counts the runs of itself.""" -from typing import Any - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import AgentCollection, EnvCollection, FlowParams, flow -@flow(resumable=True) -def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None: - state["rounds"] = state.get("rounds", 0) + 1 - agents[0].new()(f"echo round-{state['rounds']}") -''' - -#: One that takes a config as well, so the state is the argument after it. -CONFIGURED = '''"""Counts, and takes a setting.""" +class Params(FlowParams): + fail: bool = False -from typing import Any -from pydantic import BaseModel +@flow(agents=AgentCollection, envs=EnvCollection, params=Params, resumable=True) +async def counts(task, *, agents, envs, params, ctx): + ctx.state["runs"] = (ctx.state["runs"] if "runs" in ctx.state else 0) + 1 + ctx.state["resumed"] = ctx.resumed + if params.fail: + raise RuntimeError("stopped halfway") + return ctx.state["runs"] -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +@flow(agents=AgentCollection, envs=EnvCollection, params=FlowParams, resumable=True) +async def calls(task, *, agents, envs, params, ctx): + ctx.state["mine"] = "outer" + return await inner(task, agents={}, envs={}, params=FlowParams()) -class Config(BaseModel): - """What it takes.""" - step: int = 1 +@flow(agents=AgentCollection, envs=EnvCollection, params=FlowParams, resumable=True) +async def inner(task, *, agents, envs, params, ctx): + ctx.state["seen"] = (ctx.state["seen"] if "seen" in ctx.state else 0) + 1 + return ctx.state["seen"] -@flow(resumable=True) -def run( - agents: tuple[AgentBase], - task: str, - config: Config | None = None, - state: dict[str, Any] | None = None, -) -> None: - held = state if state is not None else {} - held["at"] = held.get("at", 0) + (config or Config()).step +@flow(agents=AgentCollection, envs=EnvCollection, params=FlowParams) +async def once(task, *, agents, envs, params, ctx): + return ctx.state ''' - -def _state(epic: Path) -> dict[str, object]: - """What one epic's state file holds, by flow.""" - return json.loads((epic / STATE).read_text()) +#: What every run here may spend, which is nothing it will reach. +BUDGET = {"cost": 1} -def test_a_resumable_flow_is_handed_what_the_last_run_of_it_left( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Which is the whole of it: run it again, and it goes on rather than starting over.""" +@pytest.fixture +def counts(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> str: + """The flow above, written into this project's own flows; its name.""" monkeypatch.chdir(tmp_path) - flow = written(tmp_path, "counts", COUNTS) + written(tmp_path / ".humanize" / "flows", "counts", COUNTS) + return "counts" - for _ in range(3): - Runner(flow, [ShellAgent(CONFIG)]).run("go") - first, second, third = epics() - assert _state(first) == {str(flow): {"rounds": 1}} - assert _state(second) == {str(flow): {"rounds": 2}} - assert _state(third) == {str(flow): {"rounds": 3}} - # Each run is a run of its own, whatever it picked up: an epic is never reopened. - assert read(third) is not None - assert read(third).resumable # pyright: ignore[reportOptionalMemberAccess] +def _run(named: str, *, resume: bool | Path = False, fail: bool = False) -> object: + return Runner( + named, + params={"fail": fail} if named == "counts" else None, + budget=BUDGET, + resume=resume, + ).run("go") -def test_a_run_says_which_run_it_was_picked_up_from( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """An epic is what a run was, and being the second half of another one is part of that.""" - monkeypatch.chdir(tmp_path) - flow = written(tmp_path, "counts", COUNTS) +def test_a_resumable_flow_is_handed_what_the_run_it_picks_up_left(counts: str) -> None: + assert _run(counts) == 1 + assert _run(counts, resume=True) == 2 + assert _run(counts, resume=True) == 3 - Runner(flow, [ShellAgent(CONFIG)]).run("go") - Runner(flow, [ShellAgent(CONFIG)]).run("go") + newest = epics()[-1] + assert state(newest) == {"runs": 3, "resumed": True} + assert picks_up(newest) - first, second = epics() - began = json.loads((second / "epic.jsonl").read_text().splitlines()[0]) - assert began["picked_up"] == first.name - assert began["resumable"] is True +def test_without_resume_every_run_starts_from_the_top(counts: str) -> None: + """Picking a run up is asked for; a run that was not asked to is a run of its own.""" + assert _run(counts) == 1 + assert _run(counts) == 1 -def test_a_run_picked_up_from_a_named_epic_takes_that_one_s_state( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Which is what choosing one in `/epics` and carrying on comes to.""" - monkeypatch.chdir(tmp_path) - flow = written(tmp_path, "counts", COUNTS) + assert [state(one) for one in epics()] == [ + {"runs": 1, "resumed": False}, + {"runs": 1, "resumed": False}, + ] - for _ in range(3): - Runner(flow, [ShellAgent(CONFIG)]).run("go") - first, _, _ = epics() - Runner(flow, [ShellAgent(CONFIG)], resume=first).run("go") +def test_a_run_says_which_run_it_was_picked_up_from(counts: str) -> None: + _run(counts) + first = epics()[-1] - assert _state(epics()[-1]) == {str(flow): {"rounds": 2}} + _run(counts, resume=True) + ran = read(epics()[-1]) + assert ran is not None + assert ran.picked_up == first.name + assert ran.resumable + # And the run picked up keeps what it wrote: a closed epic is never reopened. + assert state(first) == {"runs": 1, "resumed": False} -def test_a_flow_that_says_nothing_is_run_from_the_top_every_time( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Which is what every flow was before there was such a thing as picking one up.""" - monkeypatch.chdir(tmp_path) - flow = written( - tmp_path, - "plain", - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - ' agents[0].new()("echo one")\n', - ) - - Runner(flow, [ShellAgent(CONFIG)]).run("go") - - assert not resumes(flow) - assert not (epics()[0] / STATE).exists() - assert read(epics()[0]) is not None - assert not read(epics()[0]).resumable # pyright: ignore[reportOptionalMemberAccess] - - -def test_the_state_of_a_run_that_was_stopped_is_there_to_be_picked_up( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """The point of it: state saved only at the end is state a stopped run has none of.""" - monkeypatch.chdir(tmp_path) - flow = written( - tmp_path, - "stops", - '"""Writes, and then is stopped where it stands."""\n\n' - "from typing import Any\n\n" - "from hmz.coganchor.agents import AgentBase, Stopped\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow(resumable=True)\n" - "def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None:\n" - ' state["reached"] = "half way"\n' - ' raise Stopped("esc")\n', - ) - - with pytest.raises(Stopped): - Runner(flow, [ShellAgent(CONFIG)]).run("go") - (epic,) = epics() - assert state(epic) == {"reached": "half way"} - assert resumed(str(flow)) == epic +def test_a_run_picked_up_from_a_named_epic_takes_that_one_s_journal( + counts: str, +) -> None: + _run(counts) + first = epics()[-1] + _run(counts, resume=True) + _run(counts, resume=True) + assert _run(counts, resume=first) == 2 -def test_something_written_inside_the_state_is_saved_when_the_run_ends( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A list appended to is a change no mapping can see, and is still what the flow kept.""" - monkeypatch.chdir(tmp_path) - flow = written( - tmp_path, - "appends", - '"""Appends to a list it keeps."""\n\n' - "from typing import Any\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow(resumable=True)\n" - "def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None:\n" - ' state.setdefault("seen", []).append(task)\n', - ) - - Runner(flow, [ShellAgent(CONFIG)]).run("one") - Runner(flow, [ShellAgent(CONFIG)]).run("two") - - assert state(epics()[-1]) == {"seen": ["one", "two"]} - - -def test_a_resumable_flow_that_takes_a_config_is_handed_both( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """The state is the argument after the config, which is where the flow declared it.""" - monkeypatch.chdir(tmp_path) - flow = written(tmp_path, "configured", CONFIGURED) - Runner(flow, [ShellAgent(CONFIG)], {"step": 4}).run("go") - Runner(flow, [ShellAgent(CONFIG)], {"step": 4}).run("go") +def test_what_a_run_that_failed_kept_is_there_to_be_picked_up(counts: str) -> None: + """A state write is flushed as it is made, so a run that died still wrote it.""" + with pytest.raises(RuntimeError, match="halfway"): + _run(counts, fail=True) - assert state(epics()[-1]) == {"at": 8} + failed = epics()[-1] + assert (failed / RESUME).is_file() + assert resumed("counts:counts") == failed + assert _run(counts, resume=True) == 2 -def test_a_called_flow_keeps_its_own_state_under_its_own_name( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A flow that called another is two flows, and neither writes the other's.""" - monkeypatch.chdir(tmp_path) - where = tmp_path / ".humanize/flows" - where.mkdir(parents=True) - written(where, "inner", COUNTS) - written( - where, - "outer", - '"""Calls the one that counts, and counts itself."""\n\n' - "from typing import Any\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n" - "from hmz._legacy_flows import load\n\n\n" - "@flow(resumable=True)\n" - "def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None:\n" - ' state["outer"] = state.get("outer", 0) + 1\n' - ' load("inner")(agents, task)\n', - ) - - Runner("outer", [ShellAgent(CONFIG)]).run("go") - Runner("outer", [ShellAgent(CONFIG)]).run("go") - - held = _state(epics()[-1]) - assert held == {"outer": {"outer": 2}, "inner": {"rounds": 2}} - - -def test_a_flow_called_outside_a_run_is_handed_a_dict_that_is_nowhere( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch +def test_the_newest_run_that_can_be_picked_up_is_the_one_picked_up( + counts: str, ) -> None: - """A call is a call: a flow with nowhere to keep its state runs and keeps none.""" - from hmz._legacy_flows import load + """Not a run of another flow, and not one that was not resumable.""" + _run(counts) + wanted = epics()[-1] + _run("counts:once") + _run("counts:calls") - monkeypatch.chdir(tmp_path) - where = tmp_path / ".humanize/flows" - where.mkdir(parents=True) - written(where, "counts", COUNTS) + assert resumed("counts:counts") == wanted + assert resumed("counts") == wanted # as it was named, as well as by its ref + assert resumed("counts:once") is None - load("counts")([ShellAgent(CONFIG)], "go") - assert epics() == [] +def test_a_called_flow_keeps_its_own_state_under_its_own_name(counts: str) -> None: + """A resumable run picks up every call it made that is made the same way again.""" + assert _run("counts:calls") == 1 + assert _run("counts:calls", resume=True) == 2 + newest = epics()[-1] + assert state(newest) == {"mine": "outer"} + assert state(newest, "counts:inner") == {"seen": 2} -def test_a_flow_that_says_it_resumes_and_takes_no_dict_says_so( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A flow is called with what it declares, so one that declares neither is one to correct.""" - monkeypatch.chdir(tmp_path) - flow = written( - tmp_path, - "short", - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow(resumable=True)\n" - "def run(agents: tuple[AgentBase], task: str) -> None:\n" - " pass\n", - ) - - assert resumes(flow) - with pytest.raises(TypeError): - Runner(flow, [ShellAgent(CONFIG)]).run("go") - - -def test_asking_a_flow_that_is_not_one_whether_it_resumes_says_it_is_not_one( - tmp_path: Path, -) -> None: - """Read by running the flow, so a name nothing answers to is refused as ever.""" - del tmp_path - with pytest.raises(NotAFlow): - resumes("no_such_flow_anywhere") + +def test_a_flow_that_is_not_resumable_is_not_picked_up(counts: str) -> None: + with pytest.raises(Refused, match="does not say it can be picked up"): + Runner("counts:once", budget=BUDGET, resume=True) -def test_a_flow_that_emptied_its_state_starts_the_next_run_clean( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch +def test_a_run_with_nothing_to_pick_up_says_so(counts: str) -> None: + with pytest.raises(Refused, match="no run here to pick up"): + Runner("counts", budget=BUDGET, resume=True) + + +def test_a_flow_that_is_not_resumable_keeps_no_journal_and_no_state( + counts: str, ) -> None: - """Clearing it says the next run starts clean, and is not the same as never writing. + assert _run("counts:once") is None - A run that wrote nothing at all is nothing to pick up and the search goes past it. A run - that wrote and then emptied what it had written is a run that finished with nothing to - hand on -- and handing the next one the state of the run before that would be answering - the opposite of what it said. - """ - monkeypatch.chdir(tmp_path) - flow = written( - tmp_path, - "clears", - '"""Counts, and clears what it kept when it is told to stop counting."""\n\n' - "from typing import Any\n\n" - "from hmz.coganchor.agents import AgentBase\n" - "from hmz._legacy_flows import flow\n\n\n" - "@flow(resumable=True)\n" - "def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None:\n" - ' if task == "done":\n' - " state.clear()\n" - " return\n" - ' state["rounds"] = state.get("rounds", 0) + 1\n', - ) - - Runner(flow, [ShellAgent(CONFIG)]).run("go") - Runner(flow, [ShellAgent(CONFIG)]).run("go") - assert state(epics()[-1]) == {"rounds": 2} - - Runner(flow, [ShellAgent(CONFIG)]).run("done") # which empties it - Runner(flow, [ShellAgent(CONFIG)]).run("go") - - # From nothing, rather than from the two rounds two runs ago. - assert state(epics()[-1]) == {"rounds": 1} + (epic,) = epics() + assert not (epic / RESUME).exists() + assert not picks_up(epic) diff --git a/tests/integration/runtime/test_run_budget.py b/tests/integration/runtime/test_run_budget.py index 05075e2a..ed459c22 100644 --- a/tests/integration/runtime/test_run_budget.py +++ b/tests/integration/runtime/test_run_budget.py @@ -1,10 +1,11 @@ -"""A run of a flow that never stops on its own, stopped by the allowance it was given. +"""A run of a flow that never stops on its own, stopped by the budget it was given. End to end and out of process: `hmz exec` on a flow whose loop has no exit of its own, under -a stand-in CLI, with a budget in the file `-c` names. What is proved is the whole of what the -allowance is for -- that the process exits rather than looping for a week, that the epic says -the run was stopped rather than done, and that each of the three dimensions does it on its -own. A unit test can prove the reckoning; only this can prove the loop actually ends. +a stand-in CLI, with the budget `-b` says. What is proved is the whole of what a budget is for +-- that the process exits rather than looping for a week, that the epic says the run was +stopped rather than done, and that each of the three dimensions does it on its own. The +engine's own tests prove the reckoning; only this can prove the loop actually ends. And a run +given no budget at all is not started: `-b` is required, so there is no run nothing will stop. """ from __future__ import annotations @@ -42,16 +43,25 @@ #: A flow whose loop has no way out at all. Every exit it could have had is deliberately #: absent, so that anything which ends this run is the run's allowance and nothing else. FOREVER = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow -@flow -def run(agents: tuple[AgentBase], task: str) -> None: +class Agents(AgentCollection): + worker: Agent + + +class Envs(EnvCollection): + here: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def forever(task, *, agents, envs, params, ctx): + session = await agents["worker"].spawn(env=envs["here"]) at = 0 while True: at += 1 - print(f"round {at}: {agents[0](task)}") + said = await agents["worker"].run(task, session=session) + print(f"round {at}: {said}", flush=True) """ @@ -102,12 +112,12 @@ def _ran( model: str = "m", timeout: float = 120.0, ) -> subprocess.CompletedProcess[str]: - """One `hmz exec` of the endless flow, under the budget written in a file. + """One `hmz exec` of the endless flow, under the budget `-b` says. Args: - tmp_path: Where the flow and the file go. + tmp_path: Where the flow goes. said: The environment, out of the `stand_in` fixture. - budget: What the file `-c` names says about what the run may spend. + budget: What `-b` says the run may spend, or "" for no `-b` at all. model: What the stand-in CLI is told to run -- `m`, which the list beside it prices, unless a test wants one nobody prices. timeout: How long to give it before it is killed, for a run that never ends. @@ -115,7 +125,6 @@ def _ran( Returns: What the process did. """ - (tmp_path / "b.yaml").write_text(budget, encoding="utf-8") return subprocess.run( [ sys.executable, @@ -125,9 +134,8 @@ def _ran( "-f", str(tmp_path / "flows" / "forever"), "-a", - f"opencode/{model}:high", - "-c", - str(tmp_path / "b.yaml"), + f"worker=opencode/{model}:high", + *(["-b", budget] if budget else []), "go", ], capture_output=True, @@ -157,16 +165,15 @@ def _how(said: dict[str, str]) -> list[str]: @pytest.mark.parametrize( ("budget", "why"), [ - # An hour a hundredth of a second long, which a loop of twenty-millisecond rounds - # reaches in the first of them. - ("budget:\n hours: 0.000003\n", "h spent"), - # Two rounds' worth of output tokens, spelled in the millions this is counted in. - (f"budget:\n tokens: {2 * EACH / 1_000_000}\n", "output tokens spent"), + # A twentieth of a second, which a loop of fifty-millisecond rounds reaches at once. + ("duration=0.05", "duration"), + # Two rounds' worth of output tokens. + (f"output_tokens={2 * EACH}", "output tokens"), # And two rounds' worth of money, at the five dollars a million the list above says. - (f"budget:\n dollars: {2 * EACH * 5 / 1_000_000}\n", "$0.04 spent"), + (f"cost={2 * EACH * 5 / 1_000_000}", "cost"), ], ) -def test_a_loop_with_no_exit_of_its_own_is_stopped_by_its_allowance( +def test_a_loop_with_no_exit_of_its_own_is_stopped_by_its_budget( tmp_path: Path, stand_in: dict[str, str], budget: str, why: str ) -> None: """The whole of what this is for: a flow that would otherwise run until somebody killed it. @@ -176,27 +183,24 @@ def test_a_loop_with_no_exit_of_its_own_is_stopped_by_its_allowance( """ ran = _ran(tmp_path, stand_in, budget) - assert why in ran.stderr + ran.stdout, ran.stderr + assert ran.returncode == 0, ran.stderr + assert "hmz exec: stopped --" in ran.stderr + assert why in ran.stderr, ran.stderr # And the run is written down as stopped rather than as having finished what it set out # to do, because a run that ran out of money did not do what it was asked. assert _how(stand_in) == ["stopped"] -@pytest.mark.timeout(300) -def test_a_run_nothing_will_stop_says_so_and_runs_anyway( +@pytest.mark.timeout(120) +def test_a_run_given_no_budget_is_not_started( tmp_path: Path, stand_in: dict[str, str] ) -> None: - """A command line has nobody to ask, and refusing would break every unattended flow. - - So it says it plainly on the stream that is not the answer, and goes. The run here has no - exit at all, so it is killed rather than waited on -- which is the point being made. - """ - with pytest.raises(subprocess.TimeoutExpired) as went_on: - _ran(tmp_path, stand_in, "{}\n", timeout=10.0) - - said = (went_on.value.stderr or b"").decode(errors="replace") + """A line with no `-b` is a line to correct, before any agent has taken a turn.""" + ran = _ran(tmp_path, stand_in, "") - assert "nothing will stop this run" in said + assert ran.returncode == 2 + assert "is given a budget" in ran.stderr + assert _how(stand_in) == [] # nothing ran at all @pytest.mark.timeout(300) @@ -205,35 +209,28 @@ def test_a_cap_nothing_can_price_is_said_and_the_run_goes_on_anyway( ) -> None: """Fifty dollars on a model nobody lists, which is the run a benchmark actually made. - Bounded in the file and unbounded on the machine: nothing here can read the money, so - nothing here can stop the run. A command line has nobody to ask, so it says both things - -- which cap cannot be read, and that this leaves nothing holding the run -- and goes. + Bounded on the line and unbounded on the machine: nothing here can price the money, so + the cap cannot stop the run. A command line has nobody to ask, so it says so -- and goes. That it goes is the other half of the claim: a cell in a container must not sit waiting on a question, so the rounds have to be on stdout by the time it is killed. """ with pytest.raises(subprocess.TimeoutExpired) as went_on: - _ran( - tmp_path, - stand_in, - "budget:\n dollars: 50\n", - model="nobody-lists-this", - timeout=10.0, - ) + _ran(tmp_path, stand_in, "cost=50", model="nobody-lists-this", timeout=10.0) said = (went_on.value.stderr or b"").decode(errors="replace") - assert "nothing here can read dollars" in said - assert "nothing will stop this run" in said + assert "nobody lists a price for nobody-lists-this" in said assert "round 1" in (went_on.value.stdout or b"").decode(errors="replace") @pytest.mark.timeout(120) -def test_a_budget_written_as_one_number_stops_the_line_before_anything_runs( - tmp_path: Path, stand_in: dict[str, str] +@pytest.mark.parametrize("budget", ["25", "cost=-1", "tokens=5", "duration=soon"]) +def test_a_budget_that_cannot_be_read_stops_the_line_before_anything_runs( + tmp_path: Path, stand_in: dict[str, str], budget: str ) -> None: - """Which is what every flowverse loop's settings file says today, meaning millions.""" - ran = _ran(tmp_path, stand_in, "budget: 25\n") + """A bare number says nothing about which of the three it meant; the rest are typos.""" + ran = _ran(tmp_path, stand_in, budget) - assert ran.returncode != 0 - assert "rather than one number" in ran.stderr + assert ran.returncode == 2 + assert "-b" in ran.stderr assert _how(stand_in) == [] # nothing ran at all diff --git a/tests/integration/sdk/test_epics.py b/tests/integration/sdk/test_epics.py index 006e02b2..ff3d03dd 100644 --- a/tests/integration/sdk/test_epics.py +++ b/tests/integration/sdk/test_epics.py @@ -6,8 +6,8 @@ gathered out of a run afterwards -- a trace of it, an archive of it -- come out of the run's own directory rather than out of whatever directory somebody was standing in. -The run is a real one: a flow, driven by a stand-in agent that opens a session and echoes into -it. Nothing here starts a coding agent. +The run is a real one: a flow, driven by the engine's fake agent, which opens a session and +answers into it. Nothing here starts a coding agent. """ from __future__ import annotations @@ -17,31 +17,33 @@ import pytest -from hmz.coganchor.agents import AgentConfig -from hmz.runtime.runner import Runner -from hmz.sdk import Epics, Hmz -from tests.stubs import ShellAgent, written +from hmz.sdk import Epics, Hmz, fakes +from tests.stubs import written if TYPE_CHECKING: from pathlib import Path -CONFIG = AgentConfig(model="m", effort="high") - #: A flow that opens one session and says something into it, so that the run has one to point #: at -- which is the whole of what an epic is for -- and counts the runs of itself, so that #: the run also has something to be picked up from. FLOW = '''"""Opens a session, and counts the runs of itself.""" -from typing import Any +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow + + +class Agents(AgentCollection): + actor: Agent + -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +class Envs(EnvCollection): + here: LocalEnv -@flow(resumable=True) -def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None: - state["rounds"] = state.get("rounds", 0) + 1 - agents[0].new()("echo the-session") +@flow(agents=Agents, envs=Envs, params=FlowParams, resumable=True) +async def counts(task, *, agents, envs, params, ctx): + ctx.state["rounds"] = (ctx.state["rounds"] if "rounds" in ctx.state else 0) + 1 + session = await agents["actor"].spawn(env=envs["here"]) + await agents["actor"].run(task, session=session) ''' @@ -50,15 +52,20 @@ def ran(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: """One run of one flow in a workspace of its own, and the epic it wrote.""" monkeypatch.chdir(tmp_path) written(tmp_path, "flow", FLOW) - Runner(tmp_path / "flow", [ShellAgent(CONFIG, name="actor")]).run("go") + Hmz().run( + tmp_path / "flow", + "go", + agents={"actor": fakes.FakeAgentDriver()}, + budget={"cost": 1}, + ).run() (epic,) = Hmz().epics.all() return epic @pytest.fixture -def named(tmp_path: Path) -> str: - """The flow, as it was named when it ran, which is what its state is written under.""" - return str(tmp_path / "flow") +def named() -> str: + """The flow by its canonical ref, which is what its state is written under.""" + return "flow:counts" def test_the_runs_are_kept_under_the_directory_this_says_they_are(ran: Path) -> None: @@ -105,10 +112,11 @@ def test_what_each_agent_opened_is_under_the_name_the_run_knew_it_as(ran: Path) def test_the_last_run_of_a_flow_here_is_what_a_resumable_flow_picks_up( ran: Path, named: str ) -> None: - """Looked for by what the state holds, so a flow called by another is picked up too.""" + """The newest run of that flow here that wrote its journal.""" held = Hmz().epics assert held.resumed(named) == ran + assert held.picks_up(ran) assert held.resumed("a-flow-nobody-ran") is None diff --git a/tests/integration/sdk/test_flows.py b/tests/integration/sdk/test_flows.py index 32101127..ea9da67a 100644 --- a/tests/integration/sdk/test_flows.py +++ b/tests/integration/sdk/test_flows.py @@ -18,7 +18,7 @@ import pytest -from hmz._legacy_flows import NotAFlow +from hmz.flows import FlowNotFound from hmz.runtime.flowing import ENTRY, FLOWS, LOCAL, OFFICIAL, USER from hmz.sdk import Hmz from tests.stubs import written @@ -29,39 +29,40 @@ #: A flow, as short as one can be, that says a line about itself and takes one agent. FLOW = '''"""A flow of somebody else's.""" -from hmz._legacy_flows import Agent, flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - agent.new()(task) +class Agents(AgentCollection): + agent: Agent + + +class Envs(EnvCollection): + here: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def run(task, *, agents, envs, params, ctx): + session = await agents["agent"].spawn(env=envs["here"]) + await agents["agent"].run(task, session=session) ''' -#: One that says it can be picked up where the last run of it left off, and takes a setting. +#: One that says it can be picked up where the last run of it left off, and takes a param. KEEPS = '''"""A flow that is picked up.""" -from typing import Any - -from pydantic import BaseModel +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, flow -from hmz._legacy_flows import Agent, flow +class Agents(AgentCollection): + agent: Agent -class Config(BaseModel): - """What it takes.""" +class Params(FlowParams): rounds: int = 1 -@flow(resumable=True) -def run( - agents: tuple[Agent], - task: str, - config: Config | None = None, - state: dict[str, Any] | None = None, -) -> None: - (agent,) = agents +@flow(agents=Agents, envs=EnvCollection, params=Params, resumable=True) +async def run(task, *, agents, envs, params, ctx): + pass ''' @@ -224,25 +225,23 @@ def test_the_line_a_flow_says_about_itself_is_read_off_the_flow(project: Path) - assert Hmz().flows.about("mine") == "A flow of somebody else's." -def test_every_agent_a_flow_needs_chosen_is_read_off_how_it_declared_them( - project: Path, -) -> None: - places = Hmz().flows.places("mine") +def test_what_a_flow_declares_is_read_off_the_flow(project: Path) -> None: + """Its roles -- the ones the runtime fills marked -- its params, and whether it resumes.""" + flows = Hmz().flows + + mine = flows.declared("mine") - assert len(places) == 1 - assert not places[0].person + assert [(one.name, one.auto) for one in mine.agents] == [("agent", False)] + assert [(one.name, one.auto) for one in mine.envs] == [("here", True)] + assert "rounds" in flows.declared("kept").params.model_fields + assert mine.params.model_fields == {} -def test_what_a_flow_can_be_set_up_with_is_its_own_model_and_none_for_one_that_takes_none( +def test_a_flow_that_is_not_there_is_said_to_be_as_the_flow_api_says_it( project: Path, ) -> None: - flows = Hmz().flows - - model = flows.configures("kept") - - assert model is not None - assert "rounds" in model.model_fields - assert flows.configures("mine") is None + with pytest.raises(FlowNotFound): + Hmz().flows.declared("definitely-not-a-flow") def test_whether_a_flow_can_be_picked_up_is_what_the_flow_said(project: Path) -> None: @@ -291,53 +290,3 @@ def test_the_places_flows_come_from_are_reached_from_the_flows_as_well() -> None assert [one.name for one in held.flows.verses.all()] == [ one.name for one in held.verses.all() ] - - -# ------------------------------------------------------- what a flow is set up with - - -def test_what_a_flow_is_set_up_with_is_read_out_of_the_file_it_was_written_in( - tmp_path: Path, -) -> None: - said = tmp_path / "setup.yml" - said.write_text("rounds: 3\nname: mine\n", encoding="utf-8") - - assert Hmz().flows.set_up_from(said) == ({"rounds": 3, "name": "mine"}, None) - - -def test_a_setup_file_that_is_empty_sets_nothing_up(tmp_path: Path) -> None: - said = tmp_path / "setup.yml" - said.write_text("", encoding="utf-8") - - assert Hmz().flows.set_up_from(said) == (None, None) - - -def test_a_setup_file_that_is_not_a_mapping_is_refused(tmp_path: Path) -> None: - said = tmp_path / "setup.yml" - said.write_text("- one\n- two\n", encoding="utf-8") - - with pytest.raises(ValueError, match="mapping"): - Hmz().flows.set_up_from(said) - - -# ------------------------------------------------------------- reading a flow first - - -def test_a_flow_that_will_run_is_read_and_nothing_is_found(project: Path) -> None: - assert Hmz().flows.check("mine") == () - - -def test_the_reading_that_executes_nothing_is_the_one_that_was_asked_for( - project: Path, -) -> None: - """`static` is the whole of what it keeps: pure `ast`, and the flow is never loaded.""" - assert Hmz().flows.check("mine", static=True) == () - - -def test_a_flow_that_is_not_an_atlas_compiles_to_no_prophecy(project: Path) -> None: - assert Hmz().flows.prophecy("mine") is None - - -def test_a_flow_that_is_not_an_atlas_has_no_prophecy_to_ship(project: Path) -> None: - with pytest.raises(NotAFlow, match="not an atlas"): - Hmz().flows.foretell("mine") diff --git a/tests/integration/sdk/test_hmz.py b/tests/integration/sdk/test_hmz.py index 41508018..873bce49 100644 --- a/tests/integration/sdk/test_hmz.py +++ b/tests/integration/sdk/test_hmz.py @@ -20,15 +20,22 @@ import pytest FLOW = """ -from hmz._legacy_flows import Agent, flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, flow -@flow() -def run(agents: tuple[Agent], task: str) -> None: - (one,) = agents - print(f"ran {task}") +class Agents(AgentCollection): + one: Agent + + +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def run(task, *, agents, envs, params, ctx): + print(f"ran {task} on {agents['one'].harness}") + return task """ +#: What every line here runs its one agent as, and what it may spend. +LINE = ["-a", "one=claude/model:high", "-b", "cost=1"] + def test_the_workspace_is_the_one_it_was_given(tmp_path: pathlib.Path) -> None: assert Hmz(tmp_path).workspace == tmp_path @@ -95,9 +102,9 @@ def test_a_flow_is_run_and_says_it_ran( written.write_text(FLOW, encoding="utf-8") held = Hmz() - held.exec(["-f", str(written), "-a", "claude/model:high", "go"]) + assert held.exec(["-f", str(written), *LINE, "go"]) == "go" - assert "ran go" in capsys.readouterr().out + assert "ran go on claude" in capsys.readouterr().out def test_a_run_is_started_and_waited_for( @@ -108,19 +115,19 @@ def test_a_run_is_started_and_waited_for( written = tmp_path / "one.py" written.write_text(FLOW, encoding="utf-8") held = Hmz() - flow, agents, task, config, budget, _ = held.read( - ["-f", str(written), "-a", "claude/model:high", "go"] - ) - running = held.run(flow, agents, task, config, budget=budget) + line = held.read(["-f", str(written), *LINE, "go"]) + running = held.run(line.flow, line.task, agents=line.agents, budget=line.budget) assert not running.running + assert running.epic is None running.start() assert running.wait(timeout=30) assert running.raised is None - # The person the flow talks to is among them where it talks to one, and the agents it - # was given are the rest. - assert len(running.agents) == 1 + assert running.result == "go" + assert running.epic is not None + # And it opened nothing: the flow took no turn. + assert running.agents == () def test_the_runs_of_a_workspace_are_the_ones_run_there( @@ -130,7 +137,7 @@ def test_the_runs_of_a_workspace_are_the_ones_run_there( written = tmp_path / "one.py" written.write_text(FLOW, encoding="utf-8") held = Hmz() - held.exec(["-f", str(written), "-a", "claude/model:high", "go"]) + held.exec(["-f", str(written), *LINE, "go"]) runs = held.epics.all() @@ -148,7 +155,7 @@ def test_a_workspace_that_was_named_is_the_one_the_runs_are_read_from( monkeypatch.chdir(tmp_path) written = tmp_path / "one.py" written.write_text(FLOW, encoding="utf-8") - Hmz().exec(["-f", str(written), "-a", "claude/model:high", "go"]) + Hmz().exec(["-f", str(written), *LINE, "go"]) assert Hmz(tmp_path).epics.all() assert Hmz(elsewhere).epics.all() == [] @@ -176,3 +183,12 @@ def test_a_step_between_two_places_is_the_same_store_a_command_line_walks() -> N assert [one.spec for one in held.fallbacks.all()] == ["claude/opus"] assert held.fallbacks.clear("claude/opus") assert held.fallbacks.chain("claude/opus") == ["claude/opus"] + + +def test_the_fakes_a_flow_is_tested_on_are_offered_whole() -> None: + """The flowverse tests use them, and a tool outside wants the same kit, not a copy.""" + from hmz.runtime.flowing import fakes as there + from hmz.sdk import fakes + + assert fakes is there + assert fakes.FakeAgentDriver is there.FakeAgentDriver diff --git a/tests/integration/tracing/test_profiled_runs.py b/tests/integration/tracing/test_profiled_runs.py index 1b858717..a6e3f4e2 100644 --- a/tests/integration/tracing/test_profiled_runs.py +++ b/tests/integration/tracing/test_profiled_runs.py @@ -1,9 +1,9 @@ """A run that is profiled as well as traced, end to end. -The innovation this is here for: an agent's turns and the programs those turns ran are one +The innovation this is here for: an agent's turns and the programs a run started are one document at one scale, so that `what was this run doing at 09:41` has one answer. Driven as a -real run -- a flow, an agent that starts processes, an epic -- rather than as a profile handed -to a renderer, since what is being checked is that the two halves meet at all. +real run -- a flow, a workspace it starts processes in, an epic -- rather than as a profile +handed to a renderer, since what is being checked is that the two halves meet at all. """ from __future__ import annotations @@ -13,21 +13,18 @@ import pytest -from hmz.coganchor.agents import AgentConfig from hmz.runtime.epic import TRACES, epics, opened from hmz.runtime.runner import Runner from hmz.runtime.settings import Settings from hmz.runtime.tracing.collector import collect from hmz.runtime.tracing.profile import PROFILE, read from tests.sampling import sampled -from tests.stubs import ShellAgent, written +from tests.stubs import written if TYPE_CHECKING: import pathlib -CONFIG = AgentConfig(model="m", effort="high") - -#: What the turn runs: a shell running a sleep, which is two programs, and the profile has to +#: What the run runs: a shell running a sleep, which is two programs, and the profile has to #: hold both of them. #: # : A second rather than the tenth of one it takes to say what is being checked. What reads it : is @@ -36,15 +33,21 @@ # difference between a test of the profiler and a test of the clock. SAID = "sleep 1; echo the-session" -#: A flow whose agent runs a program, which is what a turn mostly is. +#: A flow that runs a program in its workspace, which is what a turn mostly is. FLOW = f""" -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import AgentCollection, BashEnvMixin, EnvCollection, FlowParams, LocalEnv, flow + + +class Here(LocalEnv, BashEnvMixin): ... + + +class Envs(EnvCollection): + here: Here -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - agents[0].new()("{SAID}") +@flow(agents=AgentCollection, envs=Envs, params=FlowParams) +async def run(task, *, agents, envs, params, ctx): + await envs["here"].exec("{SAID}") """ @@ -66,12 +69,12 @@ def test_a_run_is_profiled_when_the_workspace_asks_for_it( """Off unless somebody says otherwise: it is a sampler running as long as the flow does.""" Settings().profiles(on=True) - Runner(workspace / "flow", [ShellAgent(CONFIG)]).run("go") + Runner(workspace / "flow", budget={"cost": 1}).run("go") (epic,) = epics() ran = read(epic / PROFILE) assert ran, "the programs the turn ran are not in the run's profile" - # The turn itself, which is a shell running a sleep: both are programs this run started. + # What it ran, which is a shell running a sleep: both are programs this run started. # The shell is named by what it was given rather than by what it is called, one system's # `/bin/sh` being another's `bash`; the sleep is called the same thing everywhere. assert "sleep" in {one.name for one in ran} @@ -83,7 +86,7 @@ def test_a_run_nobody_asked_to_profile_is_traced_and_not_profiled( workspace: pathlib.Path, ) -> None: """A sampler nobody asked for is a sampler running for the length of every run there is.""" - Runner(workspace / "flow", [ShellAgent(CONFIG)]).run("go") + Runner(workspace / "flow", budget={"cost": 1}).run("go") (epic,) = epics() assert not (epic / PROFILE).exists() @@ -96,7 +99,7 @@ def test_the_programs_and_the_sessions_are_one_document( ) -> None: """Which is the point of profiling into a trace rather than into a profile of its own.""" Settings().profiles(on=True) - Runner(workspace / "flow", [ShellAgent(CONFIG)]).run("go") + Runner(workspace / "flow", budget={"cost": 1}).run("go") (epic,) = epics() output = epic / TRACES / "one.trace.json" diff --git a/tests/integration/tui/test_agent_sheet.py b/tests/integration/tui/test_agent_sheet.py index bf2d4245..9cd9843b 100644 --- a/tests/integration/tui/test_agent_sheet.py +++ b/tests/integration/tui/test_agent_sheet.py @@ -1,10 +1,12 @@ -"""One agent of a flow, set up on one sheet reached from the page the flow's agents are on. +"""The roles of a flow, each set up on the page the flow's roles are on. -Everything an agent is is a row, and there are four of them: the CLI that takes its turns, the -account they run as, the model at an effort, and -- only where the flow said that agent may be -pointed at a machine -- where its work lands. Which is the point: an agent is one thing rather -than three questions, and changing the effort of one already set up is a row and an arrow -rather than a walk through two sheets that had nothing to say. +Every agent role is a sheet of rows: the CLI that takes its turns, the account they run as, +and the model at an effort. Which is the point: an agent is one thing rather than three +questions, and changing the effort of one already set up is a row and an arrow rather than a +walk through two sheets that had nothing to say. An environment role is a row of its own, +where the place it works is said the way `-e` says it; the roles the runtime fills -- the +person outside the run, the workspace it was started in -- are nobody's to choose, and are no +row at all. Driven headlessly, as every test of the interface is, so what is checked is where a keystroke lands rather than how it is drawn. @@ -12,6 +14,7 @@ from __future__ import annotations +import datetime import unittest.mock from typing import TYPE_CHECKING, cast @@ -19,6 +22,7 @@ from textual.widgets import Label, OptionList from hmz.coganchor.backends import Model +from hmz.flows import Budget from hmz.runtime.kept import Runs from hmz.runtime.settings import Settings from hmz.tui import Humanize @@ -26,16 +30,15 @@ _BUDGET, _SAVE, Agent, - Anchors, Catalogue, Clis, + Configures, Confirms, Flows, - Unbounded, ) from tests.integration.tui.test_app import drops, into_agent, keeps, onto, opens, rows from tests.stubs import written -from tests.tui.fixtures import until +from tests.tui.fixtures import set_up, transcript, until if TYPE_CHECKING: from pathlib import Path @@ -45,100 +48,87 @@ #: What one installed CLI looks like, for every sheet here. CLAUDE = {"claude": (Model("claude-opus-5", ("max", "high")),)} -#: A flow whose agent it says nothing about, which is one that works here and is not asked. +#: A flow of one agent role, working in the workspace it was started in -- which is a role the +#: runtime fills, and so no row -- and talking to whoever is outside it, which is another. HERE = ''' """One agent, working where the flow is.""" -from typing import NamedTuple +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import LocalEnv, Outworlder, flow -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +class Agents(AgentCollection): + """Just the one, and the person.""" + + builder: Agent + human: Outworlder -class Agents(NamedTuple): - """Just the one.""" - builder: AgentBase +class Envs(EnvCollection): + workspace: LocalEnv -@flow -def run(agents: Agents, task: str) -> None: +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def here(task: str, *, agents: Agents, envs: Envs, params: FlowParams, + ctx: FlowContext) -> None: pass ''' -#: A flow that says its agent may be pointed at a machine, which is a row of its own. -REMOTE = ''' -"""One agent, which may work anywhere it is pointed at.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Remote -from hmz._legacy_flows import flow +#: A flow with an environment of its own, which somebody says the place of. +PLACED = ''' +"""One agent, working in a place somebody names.""" +from hmz.flows import Agent, AgentCollection, Env, EnvCollection, FlowContext, FlowParams +from hmz.flows import ShellEnvMixin, flow -class Agents(NamedTuple): - """Just the one, and it moves.""" - builder: Annotated[AgentBase, Remote] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' +class Repo(Env, ShellEnvMixin): ... -#: A flow that settles the container itself, which is a machine nobody configures. -BOXED = ''' -"""One agent, in a container of the flow's own.""" -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Isolated -from hmz._legacy_flows import flow +class Agents(AgentCollection): + """Just the one.""" + builder: Agent -class Agents(NamedTuple): - """Just the one, in a box.""" - tester: Annotated[AgentBase, Isolated("python:3.12")] +class Envs(EnvCollection): + repo: Repo -@flow -def run(agents: Agents, task: str) -> None: +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def placed(task: str, *, agents: Agents, envs: Envs, params: FlowParams, + ctx: FlowContext) -> None: pass ''' -#: Two agents that may both be pointed somewhere, which is a sheet apiece. +#: Two agent roles, which is a sheet apiece. PAIR = ''' -"""Two agents, both of which may work elsewhere.""" +"""Two agents.""" -from typing import Annotated, NamedTuple +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow -from hmz.coganchor.agents import AgentBase, Remote -from hmz._legacy_flows import flow - -class Agents(NamedTuple): +class Agents(AgentCollection): """One writes, one reads.""" - builder: Annotated[AgentBase, Remote] - reviewer: Annotated[AgentBase, Remote] + builder: Agent + reviewer: Agent -@flow -def run(agents: Agents, task: str) -> None: +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def pair(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: pass ''' @pytest.fixture def flows(tmp_path: Path) -> Path: - """Puts the four flows where this project's own would be.""" + """Puts the flows where this project's own would be.""" where = tmp_path / ".humanize" / "flows" where.mkdir(parents=True) written(where, "here", HERE) - written(where, "remote", REMOTE) - written(where, "boxed", BOXED) + written(where, "placed", PLACED) written(where, "pair", PAIR) return where @@ -149,11 +139,16 @@ def _asked(app: Humanize) -> str: def _value(app: Humanize, held: str) -> str: - """What one row of the agent sheet is set to, as it is drawn.""" + """What one row of the sheet on top is set to, as it is drawn.""" listing = app.screen.query_one("#choices", OptionList) return str(listing.get_option_at_index(rows(app).index(held)).prompt) +def _said(app: Humanize) -> str: + """What the sheet on top says under its list.""" + return str(app.screen.query_one("#tuning", Label).content) + + async def _open(app: Humanize, driver: Pilot[None], flow: str) -> None: """Opens the flow menu on one flow -- which is inside it -- and then one of its agents.""" await driver.press(*f"/flow {flow}") @@ -162,6 +157,16 @@ async def _open(app: Humanize, driver: Pilot[None], flow: str) -> None: await into_agent(app, driver) +async def _budgets(app: Humanize, driver: Pilot[None], duration: str) -> None: + """Sets what a run may spend from its row on the roles page, as a duration.""" + await onto(app, driver, _BUDGET) + await driver.press("enter") + await until(lambda: isinstance(app.screen, Configures), driver) + await driver.press(*duration) + await driver.press("enter") + await until(lambda: isinstance(app.screen, Flows), driver) + + @pytest.mark.timeout(60) @unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) async def test_one_agent_is_one_sheet_of_rows_in_the_order_they_depend( @@ -171,31 +176,44 @@ async def test_one_agent_is_one_sheet_of_rows_in_the_order_they_depend( """The CLI settles the accounts and the models, so it comes above both of them.""" app = Humanize() async with app.run_test() as driver: - await _open(app, driver, "remote") + await _open(app, driver, "here") assert "builder" in _asked(app) - # Four rows and where it works, and nothing else: what it may do, the goals it may - # reach for and whether it searches the web are the flow's, and the skills it - # carries are its CLI's -- none of them is the agent's to be asked about here. - assert rows(app) == [ - "cli", - "provider", - "model", - "effort", - "where", - _SAVE, - ] + # Four rows, and nothing else: what it may do and what it can are the flow's, where + # it works is the environment's, and the skills it carries are its CLI's -- none of + # them is the agent's to be asked about here. + assert rows(app) == ["cli", "provider", "model", "effort", _SAVE] # The account nobody chose is always the first row of the list it is chosen from. assert "as local" in _value(app, "provider") +@pytest.mark.timeout(60) +@unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) +async def test_the_roles_the_runtime_fills_are_no_rows_of_the_menu( + _installed: unittest.mock.MagicMock, # noqa: PT019 -- `mock.patch` hands it over + flows: Path, +) -> None: + """The person outside the run and the workspace it starts in are nobody's to choose.""" + app = Humanize() + async with app.run_test() as driver: + await driver.press(*"/flow here") + await driver.press("enter") + await until(lambda: isinstance(app.screen, Flows), driver) + sheet = cast("Flows", app.screen) + await until(lambda: sheet._inside, driver) + + # `builder` and nothing of `human` or `workspace`. + assert rows(app) == ["0", _BUDGET, _SAVE] + assert "builder" in _value(app, "0") + + @pytest.mark.timeout(60) @unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) async def test_two_agents_are_two_rows_and_a_sheet_apiece( _installed: unittest.mock.MagicMock, # noqa: PT019 -- `mock.patch` hands it over flows: Path, ) -> None: - """Opening a flow lists what it drives, by the name the flow calls each of them.""" + """Opening a flow lists its roles, by the name the flow calls each of them.""" app = Humanize() async with app.run_test() as driver: await driver.press(*"/flow pair") @@ -253,21 +271,19 @@ async def test_explicit_saves_accept_two_agents_then_apply_the_complete_flow( await opens(app, driver, _SAVE) await until(lambda: isinstance(app.screen, Flows), driver) + await _budgets(app, driver, "1h") await onto(app, driver, _SAVE) await driver.press("enter") - # Nobody set a budget and this flow declares none, so saving asks whether a run with - # nothing at all to stop it is what was meant. It is, here. - await until(lambda: isinstance(app.screen, Unbounded), driver) - await driver.press("enter") await until(lambda: not isinstance(app.screen, Flows), driver) - chosen = [ - Runs("claude/claude-opus-5:high"), - Runs("claude/claude-opus-5:max"), - ] + chosen = { + "builder": Runs("claude/claude-opus-5:high"), + "reviewer": Runs("claude/claude-opus-5:max"), + } assert app._flow_named == "pair" assert app._models == chosen assert Settings(tmp_path).agents("pair") == chosen + assert Settings(tmp_path).budget("pair")["duration"] == "PT1H" @pytest.mark.timeout(60) @@ -292,105 +308,115 @@ async def test_explicit_flow_save_refuses_an_agent_with_no_model( await driver.pause() assert app.screen is sheet - assert "builder has no model yet" in str( - sheet.query_one("#tuning", Label).content - ) + assert "builder is not set up yet" in _said(app) @pytest.mark.timeout(60) -@pytest.mark.parametrize("flow", ["here", "boxed"]) @unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) -async def test_where_it_works_is_asked_only_where_the_flow_says_it_moves( +async def test_a_flow_is_not_saved_until_a_run_of_it_is_given_a_budget( _installed: unittest.mock.MagicMock, # noqa: PT019 -- `mock.patch` hands it over flows: Path, - flow: str, + tmp_path: Path, ) -> None: - """A place that said nothing works here; one in a container was settled by the flow.""" + """Only a flow humanize ships runs with none; every other is refused until it has one. + + And the budget is asked on the sheet a flow's params are asked on: a duration typed as + `-b` takes one, and one that does not read is refused where it is typed. + """ app = Humanize() async with app.run_test() as driver: - await _open(app, driver, flow) - - if flow == "here": - # Nothing to say: an agent that works where the flow does is what every agent - # nobody said anything about has always been. - assert "where" not in rows(app) - else: - # Read rather than opened: the flow settled it, so nobody is being asked. - assert "in a container of python:3.12" in _value(app, "where") - await opens(app, driver, "where") - await driver.pause() - assert isinstance(app.screen, Agent) - assert "the flow settled" in str( - app.screen.query_one("#tuning", Label).content - ) + await driver.press(*"/flow here") + await driver.press("enter") + await until(lambda: isinstance(app.screen, Flows), driver) + sheet = cast("Flows", app.screen) + await until(lambda: sheet._inside, driver) + assert "none yet" in _value(app, _BUDGET) + await onto(app, driver, _SAVE) + await driver.press("enter") + await driver.pause() + assert app.screen is sheet + assert "given a budget" in _said(app) -@pytest.mark.timeout(60) -@unittest.mock.patch( - "hmz.tui.pick.machines", return_value=[("ssh://box", "ssh config")] -) -@unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) -async def test_where_an_agent_works_rides_along_with_what_it_runs( - _installed: unittest.mock.MagicMock, # noqa: PT019 -- `mock.patch` hands it over - _machines: unittest.mock.MagicMock, # noqa: PT019 - flows: Path, - tmp_path: Path, -) -> None: - """It is a setting of the agent, so it is kept beside the model and read back with it.""" - app = Humanize() - async with app.run_test() as driver: - await _open(app, driver, "remote") - await opens(app, driver, "where") - await until(lambda: isinstance(app.screen, Anchors), driver) - listing = app.screen.query_one("#choices", OptionList) - await until(lambda: bool(listing.options), driver) - # This machine first, then the ones there are to be found. - assert rows(app) == ["", "ssh://box"] - - await driver.press("down") + await onto(app, driver, _BUDGET) await driver.press("enter") - await until(lambda: isinstance(app.screen, Agent), driver) + await until(lambda: isinstance(app.screen, Configures), driver) + assert rows(app) == ["duration", "cost", "output_tokens", "graceful"] + await driver.press(*"soon") + await driver.press("enter") + await driver.pause() + assert isinstance(app.screen, Configures) # not a duration, so not taken + assert "not a duration" in _said(app) + for _ in "soon": + await driver.press("backspace") + await driver.press(*"90m") + await driver.press("enter") + await until(lambda: app.screen is sheet, driver) + assert "stops at 1h30m" in _value(app, _BUDGET) - await keeps(app, driver) - await keeps(app, driver) + await onto(app, driver, _SAVE) + await driver.press("enter") + await until(lambda: not isinstance(app.screen, Flows), driver) - chosen = Runs("claude/claude-opus-5:high", "ssh://box") - assert app._models == [chosen] - assert Settings(tmp_path).agents("remote") == [chosen] - # And a second interface opens on what this workspace was left set up to run. - again = Humanize() - assert again._models == [chosen] + assert app._budget == Budget(duration=datetime.timedelta(minutes=90)) + assert Settings(tmp_path).budget("here") == { + "duration": "PT1H30M", + "cost": None, + "output_tokens": None, + "graceful": True, + } @pytest.mark.timeout(60) -@unittest.mock.patch("hmz.tui.pick.machines", return_value=[]) @unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) -async def test_a_machine_nothing_here_can_see_is_a_target_that_is_typed( +async def test_an_environment_role_is_a_row_where_its_place_is_said( _installed: unittest.mock.MagicMock, # noqa: PT019 -- `mock.patch` hands it over - _machines: unittest.mock.MagicMock, # noqa: PT019 flows: Path, + tmp_path: Path, ) -> None: - """The list is a convenience; a target is a string, and any string that reads as one goes.""" + """Written as `-e` writes it, and read the way `-e` is: one that does not read is refused.""" app = Humanize() async with app.run_test() as driver: - await _open(app, driver, "remote") - await opens(app, driver, "where") - await until(lambda: isinstance(app.screen, Anchors), driver) - listing = app.screen.query_one("#choices", OptionList) - await until(lambda: bool(listing.options), driver) - - await driver.press("s") - await driver.press(*"nonsense") + await driver.press(*"/flow placed") + await driver.press("enter") + await until(lambda: isinstance(app.screen, Flows), driver) + sheet = cast("Flows", app.screen) + await until(lambda: sheet._inside, driver) + assert rows(app) == ["0", "@repo", _BUDGET, _SAVE] + assert "not said yet" in _value(app, "@repo") + + # Not said, so not saved: a run of it would be refused before it started. + await _budgets(app, driver, "1h") + await onto(app, driver, _SAVE) + await driver.press("enter") await driver.pause() - # Not a target and not a row, so there is nothing there to choose. - assert rows(app) == [] + assert "repo is not set up yet" in _said(app) + + await onto(app, driver, "@repo") + await driver.press("enter") + await until(lambda: isinstance(app.screen, Configures), driver) + await driver.press(*"nowhere") + await driver.press("enter") + await until(lambda: app.screen is sheet, driver) + assert "expected =" in _said(app) + assert "not said yet" in _value(app, "@repo") - for _ in range(len("nonsense")): + await onto(app, driver, "@repo") + await driver.press("enter") + await until(lambda: isinstance(app.screen, Configures), driver) + for _ in "nowhere": await driver.press("backspace") - await driver.press(*"docker://box") - await driver.pause() + await driver.press(*f"local@{tmp_path}") + await driver.press("enter") + await until(lambda: app.screen is sheet, driver) + assert f"local@{tmp_path}" in _value(app, "@repo") - assert rows(app) == ["docker://box"] + await onto(app, driver, _SAVE) + await driver.press("enter") + await until(lambda: not isinstance(app.screen, Flows), driver) + + assert app._envs == {"repo": f"local@{tmp_path}"} + assert Settings(tmp_path).envs("placed") == {"repo": f"local@{tmp_path}"} @pytest.mark.timeout(60) @@ -405,8 +431,8 @@ async def test_nothing_is_applied_until_the_menu_is_saved_on_the_way_out( """ app = Humanize() async with app.run_test() as driver: - was = (app._flow_named, list(app._models)) - await _open(app, driver, "remote") + was = (app._flow_named, dict(app._models)) + await _open(app, driver, "here") await onto(app, driver, "effort") await driver.press("left") # one effort down, which is a change await driver.pause() @@ -420,16 +446,18 @@ async def test_nothing_is_applied_until_the_menu_is_saved_on_the_way_out( assert (app._flow_named, app._models) == was # And the same walk saved lands the lot, flow and agent together. - await _open(app, driver, "remote") + await _open(app, driver, "here") await onto(app, driver, "effort") await driver.press("left") await driver.pause() await keeps(app, driver) + await until(lambda: isinstance(app.screen, Flows), driver) + await _budgets(app, driver, "1h") await keeps(app, driver) await until(lambda: not isinstance(app.screen, Flows), driver) - assert app._flow_named == "remote" - assert app._models == [Runs("claude/claude-opus-5:high")] + assert app._flow_named == "here" + assert app._models == {"builder": Runs("claude/claude-opus-5:high")} @pytest.mark.timeout(60) @@ -441,7 +469,7 @@ async def test_the_question_on_the_way_out_is_two_answers_and_esc( """Going back to the menu is what esc is everywhere else, so it is not a row as well.""" app = Humanize() async with app.run_test() as driver: - await _open(app, driver, "remote") + await _open(app, driver, "here") await onto(app, driver, "effort") await driver.press("left") await driver.pause() @@ -468,7 +496,7 @@ async def test_walking_out_of_an_unchanged_sheet_asks_nothing( """A walk in to look and out again is not a question anybody wants asked of them.""" app = Humanize() async with app.run_test() as driver: - await _open(app, driver, "remote") + await _open(app, driver, "here") await driver.press("escape") await until(lambda: isinstance(app.screen, Flows), driver) # Straight back, rather than through a question about a change nobody made. @@ -477,32 +505,23 @@ async def test_walking_out_of_an_unchanged_sheet_asks_nothing( @pytest.mark.timeout(60) @unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) -async def test_a_flow_that_puts_its_agent_here_refuses_one_that_was_pointed_away( +async def test_a_flow_given_no_budget_is_refused_where_it_is_started( _installed: unittest.mock.MagicMock, # noqa: PT019 -- `mock.patch` hands it over flows: Path, ) -> None: - """Which is why the row is only offered where the flow allows it. - - The refusal is the runner's, since where an agent works is the flow's to say -- and it is - a line at this prompt rather than a traceback out of a flow's own thread. - """ - from tests.tui.fixtures import transcript - + """The refusal is the runtime's, and it is a line at this prompt rather than a traceback.""" app = Humanize() async with app.run_test() as driver: - app._flow_named = "here" - app._wanted = app._places_of("here") - app._models = [Runs("claude/claude-opus-5:max", "ssh://box")] + set_up(app, "here", {"builder": Runs("claude/claude-opus-5:max")}) + app._budget = None await driver.press(*"go") await driver.press("enter") await until(lambda: "hmz:" in transcript(app), driver) said = transcript(app) - # Wrapped as the transcript wraps it, so it is read a phrase at a time. - assert "builder runs on this machine" in said - assert "cannot be pointed at one" in said + assert "given a budget" in said assert "Traceback" not in said # said at the prompt, not raised out of a thread - assert not app._agents # and nothing started + assert app._run is None # and nothing started @pytest.mark.timeout(60) @@ -514,7 +533,7 @@ async def test_the_flow_may_rule_a_backend_out_of_the_clis_offered( _installed: unittest.mock.MagicMock, # noqa: PT019 -- `mock.patch` hands it over tmp_path: Path, ) -> None: - """A CLI that cannot do what the place needs is one choosing would refuse to start on.""" + """A CLI that cannot do what the role declares is one choosing would refuse to start on.""" where = tmp_path / ".humanize" / "flows" where.mkdir(parents=True) written( @@ -523,20 +542,22 @@ async def test_the_flow_may_rule_a_backend_out_of_the_clis_offered( ''' """One agent, under a goal of its own.""" -from typing import Annotated, NamedTuple +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import GoalCommandAgentMixin, flow + -from hmz.coganchor.agents import AgentBase, Goal -from hmz._legacy_flows import flow +class Pursues(Agent, GoalCommandAgentMixin): ... -class Agents(NamedTuple): +class Agents(AgentCollection): """The one that pursues.""" - worker: Annotated[AgentBase, Goal] + worker: Pursues -@flow -def run(agents: Agents, task: str) -> None: +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def goal(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: pass ''', ) @@ -551,6 +572,51 @@ def run(agents: Agents, task: str) -> None: assert "claude" in rows(app) +@pytest.mark.timeout(60) +@unittest.mock.patch( + "hmz.tui.app.installed", + return_value=CLAUDE + | { + "codex": (Model("gpt-5.5", ("high",)),), + "opencode": (Model("anthropic/opus", ("high",)),), + }, +) +async def test_a_role_typed_as_one_harness_is_offered_that_harness_alone( + _installed: unittest.mock.MagicMock, # noqa: PT019 -- `mock.patch` hands it over + tmp_path: Path, +) -> None: + """A role declared a `CodexAgent` is Codex: every other CLI would be refused at the run.""" + where = tmp_path / ".humanize" / "flows" + where.mkdir(parents=True) + written( + where, + "codex_only", + ''' +"""One agent, and it is Codex.""" + +from hmz.flows import AgentCollection, CodexAgent, EnvCollection, FlowContext, FlowParams +from hmz.flows import flow + + +class Agents(AgentCollection): + reviewer: CodexAgent + + +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def codex_only(task: str, *, agents: Agents, envs: EnvCollection, + params: FlowParams, ctx: FlowContext) -> None: + pass +''', + ) + app = Humanize() + async with app.run_test() as driver: + await _open(app, driver, "codex_only") + await opens(app, driver, "cli") + await until(lambda: isinstance(app.screen, Clis), driver) + + assert rows(app) == ["codex"] + + @pytest.mark.timeout(60) @unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) async def test_the_models_are_what_that_cli_last_said_and_are_asked_again_on_r( @@ -560,7 +626,7 @@ async def test_the_models_are_what_that_cli_last_said_and_are_asked_again_on_r( """A CLI ships a model without asking anybody, so the list is asked for rather than kept.""" app = Humanize() async with app.run_test() as driver: - await _open(app, driver, "remote") + await _open(app, driver, "here") await opens(app, driver, "model") await until(lambda: isinstance(app.screen, Catalogue), driver) keys = str(app.screen.query_one("#keys", Label).content) diff --git a/tests/integration/tui/test_app.py b/tests/integration/tui/test_app.py index ffdc7203..69e95ef1 100644 --- a/tests/integration/tui/test_app.py +++ b/tests/integration/tui/test_app.py @@ -21,6 +21,7 @@ from hmz.coganchor.agents import DshSession from hmz.coganchor.backends import Model +from hmz.flows import Budget from hmz.runtime.epic import epics from hmz.runtime.kept import Runs from hmz.tui import Humanize @@ -39,63 +40,22 @@ Flows, Monitoring, Signing, - Unbounded, Ways, ) from tests.stubs import events as recorded from tests.stubs import written -from tests.tui.fixtures import transcript, until +from tests.tui.fixtures import ONE, holding, set_up, transcript, until if TYPE_CHECKING: from pathlib import Path from textual.pilot import Pilot -#: A flow that drives one agent for two turns, so a line can be typed while it is running. -FLOW = """ -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - session = agents[0].new() - Path("said.txt").write_text(session(task) + "\\n") -""" - -#: A flow that catches its own turn, so that a turn ended by hand can be seen for what it -#: leaves behind: what the call answered with, and whether the agent was stopped along with it. -CATCHING = """ -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - session = agents[0].new() - said = session(task, suppress=True) - Path("said.txt").write_text(f"{said!r} {agents[0].stopped}\\n") -""" - -#: The same flow written as a coroutine, which the interface runs exactly as it runs the -#: other kind: on a thread of its own, with the agents it was handed reaching back here. -AWAITED = """ -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +#: A flow of one agent for one turn, which writes down what the turn answered -- so a line +#: can be typed while it is running. +FLOW = ONE -@flow -async def run(agents: tuple[AgentBase], task: str) -> None: - session = agents[0].new() - Path("said.txt").write_text(await session.aturn(task) + "\\n") -""" - #: A `claude` that answers each thing it is told with a turn of its own, as the real one #: does, but withholds the first answer until a second thing arrives -- which is what makes #: the interjection observable: the turn cannot end before the typed line lands. @@ -239,11 +199,6 @@ async def _leaves(app: Humanize, driver: Pilot[None], *answer: str) -> None: if isinstance(app.screen, Confirms): await driver.press(*answer) await driver.pause() - # And, where saving is what was answered and the run it saves has nothing at all to stop - # it, the second question about that -- which every flow in this suite is asked, none of - # them declaring a budget and none of these tests setting one. - if isinstance(app.screen, Unbounded): - await driver.press("enter") await until(lambda: app.screen is not was, driver) @@ -294,34 +249,12 @@ async def test_a_line_typed_while_a_flow_runs_reaches_the_agent( written(workspace, "flow", FLOW) app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await driver.press(*"start") await driver.press("enter") # The turn will not end until it has been told something else, so this cannot race. await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), - driver, - ) - await driver.press(*"and this") - await driver.press("enter") - await until(lambda: bool((workspace / "said.txt").exists()), driver) - - assert (workspace / "said.txt").read_text().strip() == "start then and this" - - -@pytest.mark.timeout(60) -async def test_a_flow_that_is_a_coroutine_runs_here_as_any_other_does( - workspace: Path, -) -> None: - """A flow written as `async def run` is a flow: started, typed at, and done with here.""" - written(workspace, "flow", AWAITED) - app = Humanize() - async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] - await driver.press(*"start") - await driver.press("enter") - await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), + lambda: any(agent.sessions for agent in app._agents), driver, ) await driver.press(*"and this") @@ -340,7 +273,7 @@ async def test_a_line_with_nothing_to_run_it_on_says_so_rather_than_vanishing() await driver.press("enter") await driver.pause() - assert app._models == [] # nothing installed, so nothing was set up to run + assert app._models == {} # nothing installed, so nothing was set up to run assert "no coding agent is installed here" in transcript(app) @@ -360,7 +293,7 @@ async def test_a_flow_that_is_not_there_is_a_line_to_correct_and_not_the_end() - """A flow chosen that will not load is said so, and the interface stays up.""" app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "nowhere.py", [Runs("claude/m:high")] + set_up(app, "nowhere.py") await driver.press(*"do it") await driver.press("enter") await until(lambda: "nowhere.py" in transcript(app), driver) @@ -373,16 +306,15 @@ async def test_a_flow_that_is_not_there_is_a_line_to_correct_and_not_the_end() - async def test_a_flow_that_fails_as_it_is_read_is_a_line_to_correct_and_not_the_end( workspace: Path, ) -> None: - """Reading a flow runs it, so a flow may fail before it has been asked to do anything. + """Reading a flow imports it, so a flow may fail before it has been asked to do anything. - One that opens a file beside it and does not find it raises where the line is read for - what its agents should default to -- which is a convenience, and not somewhere an - interface may die. + One that opens a file beside it and does not find it raises as it is imported -- which + is where the run is refused, and not somewhere an interface may die. """ written(workspace, "broken", 'raise FileNotFoundError("no prompt.md beside me")\n') app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "broken", [Runs("claude/m:high")] + set_up(app, "broken") await driver.press(*"do it") await driver.press("enter") await until(lambda: "prompt.md" in transcript(app), driver) @@ -404,11 +336,11 @@ async def test_what_the_flow_did_is_on_monitor(workspace: Path) -> None: written(workspace, "flow", FLOW) app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await driver.press(*"start") await driver.press("enter") await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), + lambda: any(agent.sessions for agent in app._agents), driver, ) await driver.press(*"and this") @@ -643,11 +575,11 @@ async def test_what_is_running_is_not_swapped_underneath_itself( written(workspace, "flow", FLOW) app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await driver.press(*"start") await driver.press("enter") await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), + lambda: any(agent.sessions for agent in app._agents), driver, ) @@ -665,7 +597,7 @@ async def test_what_is_running_is_not_swapped_underneath_itself( await until(lambda: not isinstance(app.screen, Flows), driver) assert app._flow_named == "flow" # nothing got anywhere - assert app._models == [Runs("claude/m:high")] + assert app._models == {"coder": Runs("claude/m:high")} # And `/monitor` is not refused either: it is read, so nothing conflicts with it. app.action_monitor() await until(lambda: isinstance(app.screen, Monitoring), driver) @@ -674,7 +606,7 @@ async def test_what_is_running_is_not_swapped_underneath_itself( await driver.press("escape") await until(lambda: not isinstance(app.screen, Monitoring), driver) app.action_stop_flow() - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None and app._stopping is None, driver) @pytest.mark.timeout(90) @@ -682,23 +614,24 @@ async def test_what_is_running_is_not_swapped_underneath_itself( "hmz.tui.app.installed", return_value={"claude": (Model("m", ("max", "high")),)}, ) -async def test_an_agent_is_set_up_again_under_the_flow_that_is_running_it( +async def test_an_agent_set_up_under_a_running_flow_is_what_the_next_run_starts_on( _installed: unittest.mock.MagicMock, # noqa: PT019 -- `mock.patch` hands it over workspace: Path, ) -> None: - """The whole reason that page is never shut. + """That page is never shut, and what is saved there waits for the next run. - An agent thinking too little is found out halfway through a run, and stopping the flow to - fix it is not the answer. + A run is handed a driver per role as it starts, and every session of that role opens on + it: the agent under way is not swapped under the flow holding it, and saying so is the + whole of what saving does to it. """ written(workspace, "flow", FLOW) app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await driver.press(*"start") await driver.press("enter") await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), + lambda: any(agent.sessions for agent in app._agents), driver, ) (agent,) = app._agents @@ -713,22 +646,21 @@ async def test_an_agent_is_set_up_again_under_the_flow_that_is_running_it( await driver.pause() await keeps(app, driver) await keeps(app, driver) - await until(lambda: agent.config.effort == "max", driver) + await until(lambda: "the next run starts on" in transcript(app), driver) - # Said, and said about the agent rather than about the flow: what changed is what - # this one runs from its next turn on. - assert "claude/m:max" in transcript(app) - assert app._models == [Runs("claude/m:max")] + assert agent.config.effort == "high" # the running one, as it started + assert app._models == {"coder": Runs("claude/m:max")} # and the next, as saved + app.action_stop_flow() + await until(lambda: app._run is None and app._stopping is None, driver) @pytest.mark.timeout(90) async def test_two_ctrl_c_stop_the_flow_and_not_just_the_turn(workspace: Path) -> None: - """Ctrl+c twice ends the loop, rather than letting it hand on to the next agent. + """Ctrl+c twice ends the run, rather than only the turn under way. A flow is a loop, so stopping the turn under way is not stopping anything: the loop - would go round again. Every agent is told, and the one that raises `Stopped` takes the - loop with it -- which is why `Stopped` is not the failed turn a flow's own `|| true` - catches. + would go round again. The run is stopped, which interrupts the turn and unwinds every + call of the flow from where it stands. Twice, because a day's work is behind a key that is also pressed by mistake: the first press says what the next one does and the second one does it. @@ -736,20 +668,21 @@ async def test_two_ctrl_c_stop_the_flow_and_not_just_the_turn(workspace: Path) - written(workspace, "flow", FLOW) app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await driver.press(*"start") await driver.press("enter") await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), + lambda: any(agent.sessions for agent in app._agents), driver, ) await driver.press("ctrl+c") await driver.pause() - assert app._agents # one press asks, and asks rather than doing it + assert app._run is not None # one press asks, and asks rather than doing it assert "press ctrl+c again" in transcript(app) await driver.press("ctrl+c") - await until(lambda: not app._agents, driver) # the flow itself is over + await until(lambda: app._run is None, driver) # the flow itself is over + await until(lambda: app._stopping is None, driver) # and has unwound assert "stopping the flow" in transcript(app) # And the run is over with it: an epic is one run of one flow, and this ends one. @@ -771,23 +704,23 @@ async def test_escape_is_how_the_run_is_read_rather_than_how_it_is_stopped( written(workspace, "flow", FLOW) app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await driver.press(*"start") await driver.press("enter") await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), + lambda: any(agent.sessions for agent in app._agents), driver, ) - held = app._agents + held = app._run await driver.press("escape") await until(lambda: isinstance(app.screen, Monitoring), driver) - assert app._agents is held # read, and nothing stopped by reading it + assert app._run is held # read, and nothing stopped by reading it await driver.press("escape") await driver.pause() app.action_stop_flow() - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None and app._stopping is None, driver) @pytest.mark.timeout(90) @@ -804,8 +737,8 @@ async def test_a_line_to_a_running_flow_is_never_turned_away(workspace: Path) -> app = Humanize() async with app.run_test() as driver: # A flow that is running, with nobody mid-turn: an agent that has launched nothing. - app._flow_named, app._models = "flow", [Runs("claude/m:high")] - app._agents = [ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high"))] + set_up(app, "flow") + holding(app, ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high"))) app._queued = [] await driver.press(*"and this") await driver.press("enter") @@ -859,7 +792,7 @@ async def test_a_flow_between_two_turns_is_a_flow_that_is_running() -> None: async with app.run_test() as driver: # A flow that is running, with nobody mid-turn: the flow's own code has the time. app._flow_named = "rlcr" - app._agents = [ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high"))] + holding(app, ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high"))) app._monitor.began = time.monotonic() - 90 app._draw() await driver.pause() @@ -873,22 +806,21 @@ async def test_a_flow_between_two_turns_is_a_flow_that_is_running() -> None: @pytest.mark.timeout(60) async def test_a_flow_that_called_another_names_both_of_them() -> None: """A flow may reach for another and run it, and what is running is then both.""" - from hmz.runtime.flowing.driving import entered, left + from hmz.runtime.flowing import LiveCall + started = LiveCall("chat:chat", "chat", 1, time.monotonic(), 0, None) + called = LiveCall("rlar:review", "review", 2, time.monotonic(), 0, started) app = Humanize() async with app.run_test() as driver: app._flow_named = "chat" - started = entered("chat") - called = entered("official/rlar") - try: + with unittest.mock.patch.object( + type(app.hmz.flows), "running", return_value=(started, called) + ): app._draw() await driver.pause() status = str(app.query_one("#status", Static).content) - finally: - left(called) - left(started) - assert "chat ▸ official/rlar" in status + assert "chat ▸ rlar:review" in status # And back to the one that is set up to run, once nothing is. app._draw() @@ -1168,7 +1100,8 @@ async def test_a_flow_is_opened_to_reach_its_agents_and_esc_comes_back() -> None await driver.press("enter") await until(lambda: sheet._inside, driver) - # What the flow drives, and the row the lot is saved from. + # The role the person chooses an agent for -- the person being the other, and + # nobody's to choose -- what a run may spend, and the row the lot is saved from. assert rows(app) == ["0", _BUDGET, _SAVE] assert "chat" in str(sheet.query_one("#asked", Label).content) assert "esc back to the flows" in str( @@ -1183,7 +1116,7 @@ async def test_a_flow_is_opened_to_reach_its_agents_and_esc_comes_back() -> None await driver.press("escape") await until(lambda: not isinstance(app.screen, Flows), driver) - assert app._models == [] + assert app._models == {} @pytest.mark.timeout(60) @@ -1277,11 +1210,14 @@ def asking(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: @pytest.mark.timeout(90) async def test_an_agent_that_stops_to_ask_reaches_the_prompt(asking: Path) -> None: - """The point of the prompt being here at all: a turn that needs a person can have one.""" - written(asking, "flow", FLOW) + """The point of the prompt being here at all: a turn that needs a person can have one. + + Through the flow, which hangs its agent's question on whoever is outside the run -- as + `chat` does -- and so through this prompt. + """ app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "chat", {"assistant": Runs("claude/m:high")}) await driver.press(*"start") await driver.press("enter") await until(lambda: "Which way?" in transcript(app), driver) @@ -1291,12 +1227,14 @@ async def test_an_agent_that_stops_to_ask_reaches_the_prompt(asking: Path) -> No assert "right" in transcript(app) await driver.press(*"right") await driver.press("enter") - await until((asking / "said.txt").exists, driver) + await until(lambda: "updatedInput" in app._last_answer, driver) - # Which reached the tool as its answer, against the question it was asked. - assert json.loads((asking / "said.txt").read_text())["updatedInput"]["answers"] == { - "Which way?": "right" - } + # Which reached the tool as its answer, against the question it was asked. + assert json.loads(app._last_answer)["updatedInput"]["answers"] == { + "Which way?": "right" + } + app.action_stop_flow() + await until(lambda: app._run is None and app._stopping is None, driver) @pytest.mark.timeout(90) @@ -1304,10 +1242,9 @@ async def test_away_means_the_agent_is_told_nobody_is_there_rather_than_waiting( asking: Path, ) -> None: """A question nobody is going to answer is a flow that has stopped, so it is refused.""" - written(asking, "flow", FLOW) app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "chat", {"assistant": Runs("claude/m:high")}) await driver.press(*"/afk") await driver.press("enter") assert ( @@ -1316,10 +1253,12 @@ async def test_away_means_the_agent_is_told_nobody_is_there_rather_than_waiting( await driver.press(*"start") await driver.press("enter") - await until((asking / "said.txt").exists, driver) + await until(lambda: bool(app._last_answer), driver) + # And an outworlder that is away answers nothing, so the conversation is over too. + await until(lambda: app._run is None, driver) # Nobody answered, so the tool was declined and the turn carried on from that. - assert json.loads((asking / "said.txt").read_text())["behavior"] == "deny" + assert json.loads(app._last_answer)["behavior"] == "deny" @pytest.mark.timeout(60) @@ -1353,24 +1292,12 @@ async def test_a_third_ctrl_c_does_not_wait_for_the_flow_to_unwind() -> None: agent fallen over by itself. It is the last thing a key can do about a run. """ from hmz.coganchor.agents import AgentConfig, Event - from tests.stubs import ShellAgent, ShellSession - - closed: list[ShellSession] = [] - - class Closing(ShellSession): - def close(self) -> None: - closed.append(self) - super().close() - - class Closes(ShellAgent): - def new(self, cwd: object = None) -> Closing: - del cwd - return Closing(self) + from tests.stubs import ShellAgent app = Humanize() async with app.run_test() as driver: - agent = Closes(AgentConfig(model="m", effort="high")) - app._agents = [agent] + agent = ShellAgent(AgentConfig(model="m", effort="high")) + run = holding(app, agent) session = agent.new() app._heard(agent, session, Event(kind="begins", text="")) await driver.pause() @@ -1378,23 +1305,23 @@ def new(self, cwd: object = None) -> Closing: await driver.press("ctrl+c") await driver.press("ctrl+c") await driver.pause() - assert not app._agents # stopped, and still unwinding - assert app._stopping == [agent] + assert app._run is None # stopped, and still unwinding + assert app._stopping is run + assert run.stopped assert ( session in app._working ) # which is a turn nothing has reported the end of - closed.clear() await driver.press("ctrl+c") await driver.pause() - # Closed again, whatever the first close came to: a backend that ignored one is the - # reason there is a third press at all. And the run reads as over from here. - assert closed == [session] + # Closed, whatever the stop came to: a backend that ignored one is the reason there + # is a third press at all. And the run reads as over from here. + assert run.closed assert session not in app._working assert "closing 1 conversation" in transcript(app) assert app.is_running # the run, rather than the interface - assert not app._stopping # and nothing left for a fourth press to reach + assert app._stopping is None # and nothing left for a fourth press to reach @pytest.mark.timeout(60) @@ -1589,7 +1516,7 @@ async def test_it_opens_ready_to_be_talked_to(talking: Path) -> None: assert app._flow_named == "chat" # The first agent installed, at the first model it runs -- but never at the hardest # effort, which is a thing to ask for rather than to spend before anyone has. - assert app._models == [Runs("claude/claude-opus-5:high")] + assert app._models == {"assistant": Runs("claude/claude-opus-5:high")} @pytest.mark.timeout(60) @@ -1600,7 +1527,7 @@ async def test_it_opens_saying_so_when_there_is_nothing_to_talk_to() -> None: await driver.pause() assert app._flow_named == "chat" - assert app._models == [] + assert app._models == {} assert app.is_running @@ -1621,10 +1548,10 @@ async def test_every_line_typed_between_turns_is_a_turn_of_one_conversation( await driver.press("enter") await until(lambda: "heard second" in transcript(app), driver) - # One agent and the person, one session: the second turn resumed the first rather - # than opening another, so the agent had the first in context. - agent, person = app._agents - assert person.backend == "human" + # One session: the second turn was taken in the first's conversation rather than in + # another, so the agent had the first in context. + (agent,) = app._agents + assert agent.id == "assistant" assert agent.opened == [agent.opened[0]] # And nothing is left pinned above the prompt: a flow waiting to be told something # takes what was typed at once, so it was said rather than held. @@ -1632,7 +1559,7 @@ async def test_every_line_typed_between_turns_is_a_turn_of_one_conversation( assert app._queued == [] app.action_stop_flow() - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None and app._stopping is None, driver) @unittest.mock.patch( @@ -1649,7 +1576,7 @@ def test_deepseek_with_a_local_key_can_be_the_chat_default( app = Humanize() - assert app._models == [Runs("dsh/deepseek-v4-flash:high")] + assert app._models == {"assistant": Runs("dsh/deepseek-v4-flash:high")} @pytest.mark.timeout(60) @@ -1723,7 +1650,7 @@ def running(_session: DshSession) -> unittest.mock.MagicMock: "hmz.coganchor.agents.dsh.uuid.uuid4", lambda: SimpleNamespace(hex="chat") ) - app = Humanize(agents=[Runs("dsh/deepseek-v4-flash:high")]) + app = Humanize(agents={"assistant": Runs("dsh/deepseek-v4-flash:high")}) async with app.run_test() as driver: await driver.press(*"hello") await driver.press("enter") @@ -1734,7 +1661,7 @@ def running(_session: DshSession) -> unittest.mock.MagicMock: assert sent.kwargs["notification_subscription"] is subscription app.action_stop_flow() - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None and app._stopping is None, driver) @pytest.mark.timeout(90) @@ -1750,7 +1677,7 @@ async def test_a_flow_waiting_to_be_told_something_can_still_be_stopped( await driver.press("ctrl+c") await driver.press("ctrl+c") - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None and app._stopping is None, driver) # And the run is written down as stopped rather than as one that finished. (epic,) = epics(talking) @@ -1779,13 +1706,12 @@ async def test_clearing_the_screen_clears_the_screen_and_nothing_else( # The screen, and only the screen: what was set up to run is still set up to run, # and what is running is still running. assert app._flow_named == "chat" - assert app._models == [Runs("claude/claude-opus-5:high")] - assert ( - app._agents - ) # the flow the first line started was not stopped with the screen + assert app._models == {"assistant": Runs("claude/claude-opus-5:high")} + # The flow the first line started was not stopped with the screen. + assert app._run is not None app.action_stop_flow() - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None and app._stopping is None, driver) @pytest.mark.timeout(60) @@ -1795,7 +1721,7 @@ async def test_looking_at_the_flows_and_walking_out_changes_nothing( """Tab is pressed to see what there is, and seeing must not cost what was set up.""" app = Humanize() async with app.run_test() as driver: - was = (app._flow_named, list(app._models)) + was = (app._flow_named, dict(app._models)) await driver.press(*"/flow") await driver.press("enter") @@ -1971,7 +1897,7 @@ def effort() -> str: await keeps(app, driver) # out of the agent, holding it await keeps(app, driver) # and out of the menu, saving the lot - assert app._models == [Runs("claude/claude-opus-5:high")] + assert app._models == {"assistant": Runs("claude/claude-opus-5:high")} @pytest.mark.timeout(60) @@ -2016,7 +1942,7 @@ async def test_changing_the_cli_lets_go_of_the_model_that_belonged_to_the_last_o await keeps(app, driver) await keeps(app, driver) - assert app._models == [Runs("codex/gpt-5.5:high")] + assert app._models == {"assistant": Runs("codex/gpt-5.5:high")} @pytest.mark.timeout(60) @@ -2047,24 +1973,30 @@ async def test_deepseek_is_selectable_from_agents( goal = tmp_path / "goal.py" goal.write_text( """ -from typing import Annotated, NamedTuple +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import GoalCommandAgentMixin, flow + -from hmz.coganchor.agents import AgentBase, Goal -from hmz._legacy_flows import flow +class Worker(Agent, GoalCommandAgentMixin): ... -class Agents(NamedTuple): - worker: Annotated[AgentBase, Goal] +class Agents(AgentCollection): + worker: Worker -@flow -def run(agents: Agents, task: str) -> None: +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def goal(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: pass """ ) - app = Humanize(flow=str(goal), agents=[Runs("claude/claude-opus-5:max")]) + app = Humanize(flow=str(goal), agents={"worker": Runs("claude/claude-opus-5:max")}) + app._budget = Budget(cost=1) async with app.run_test() as driver: - await into_flows(app, driver) + # Opened on the flow by its path, which is a flow of nobody's list: its one role + # runs `/goal`, which DeepSeek's harness serves. + await driver.press(*f"/flow {goal}") + await driver.press("enter") await into_agent(app, driver) await opens(app, driver, "cli") @@ -2095,7 +2027,7 @@ def run(agents: Agents, task: str) -> None: await keeps(app, driver) await keeps(app, driver) - assert app._models == [Runs("dsh/deepseek-v4-pro:max")] + assert app._models == {"worker": Runs("dsh/deepseek-v4-pro:max")} @pytest.mark.timeout(60) @@ -2117,7 +2049,7 @@ async def test_deepseek_has_its_own_ways_after_switching_from_kimi( monkeypatch.delenv("DEEPSEEK_API_KEY", raising=False) providers.add("kimi", "subscription", way="login") - app = Humanize(agents=[Runs("kimi/kimi-code/k3:high")]) + app = Humanize(agents={"assistant": Runs("kimi/kimi-code/k3:high")}) async with app.run_test() as driver: await into_flows(app, driver) await into_agent(app, driver) @@ -2191,7 +2123,7 @@ async def test_deepseek_has_its_own_ways_after_switching_from_kimi( assert made is not None assert made.way == "key" assert made.env == {"DEEPSEEK_API_KEY": "test-key"} - assert app._models == [Runs("dsh/deepseek-v4-flash:max", "", "", "mine")] + assert app._models == {"assistant": Runs("dsh/deepseek-v4-flash:max", "mine")} @pytest.mark.timeout(60) @@ -2212,7 +2144,7 @@ async def test_agents_says_how_to_install_deepseek_when_its_sdk_is_missing( _installable: unittest.mock.MagicMock, # noqa: PT019 -- patch hands it over _installed: unittest.mock.MagicMock, # noqa: PT019 -- patch hands it over ) -> None: - app = Humanize(agents=[Runs("claude/claude-opus-5:max")]) + app = Humanize(agents={"assistant": Runs("claude/claude-opus-5:max")}) async with app.run_test() as driver: await into_flows(app, driver) await into_agent(app, driver) @@ -2231,7 +2163,7 @@ async def test_agents_says_how_to_install_deepseek_when_its_sdk_is_missing( await drops(app, driver) await drops(app, driver) - assert app._models == [Runs("claude/claude-opus-5:max")] + assert app._models == {"assistant": Runs("claude/claude-opus-5:max")} def test_deepseek_install_hint_targets_the_python_running_humanize( @@ -2239,7 +2171,7 @@ def test_deepseek_install_hint_targets_the_python_running_humanize( ) -> None: from hmz.tui import pick - monkeypatch.setattr(pick.sys, "executable", "/opt/hmz/bin/python") + monkeypatch.setattr("hmz.tui.pick.sys.executable", "/opt/hmz/bin/python") assert "uv pip install --python /opt/hmz/bin/python" in pick._installing("dsh") @@ -2402,7 +2334,7 @@ def shown(held: str) -> str: await keeps(app, driver) # One turn, at one effort, run wide -- which is how Kimi is asked for a fleet. - assert app._models == [Runs("kimi/kimi-code/k3:swarmlow")] + assert app._models == {"assistant": Runs("kimi/kimi-code/k3:swarmlow")} @pytest.mark.timeout(60) @@ -2624,9 +2556,7 @@ async def test_two_things_said_get_two_answers_and_not_three( await driver.press(*"first") await driver.press("enter") # The turn will not end until it has been told something else, so this cannot race. - await until( - lambda: bool(app._agents and any(a.sessions for a in app._agents)), driver - ) + await until(lambda: any(a.sessions for a in app._agents), driver) await driver.press(*"second") await driver.press("enter") await until(lambda: "answer to second" in transcript(app), driver) @@ -2637,7 +2567,7 @@ async def test_two_things_said_get_two_answers_and_not_three( assert shown.count("answer to second") == 1 app.action_stop_flow() - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None and app._stopping is None, driver) def test_the_offers_say_what_each_command_takes() -> None: @@ -2711,20 +2641,23 @@ async def test_the_box_at_the_top_says_what_this_is_and_not_what_is_set_up( QUESTIONNAIRE = ''' """Ask the person, in a shape.""" -import json from pathlib import Path -from typing import Literal, NamedTuple +from typing import Literal from pydantic import BaseModel, Field -from hmz.coganchor.agents import HumanAgent -from hmz._legacy_flows import flow +from hmz.flows import AgentCollection, EnvCollection, FlowContext, FlowParams, LocalEnv +from hmz.flows import Outworlder, flow -class Agents(NamedTuple): +class Agents(AgentCollection): """Nobody but the person.""" - human: HumanAgent + human: Outworlder + + +class Envs(EnvCollection): + workspace: LocalEnv class Settled(BaseModel): @@ -2737,10 +2670,13 @@ class Settled(BaseModel): rounds: int = Field(default=3, description="How many rounds may it take?") -@flow -def run(agents: Agents, task: str) -> None: - settled = agents.human(task, schema=Settled, suppress=True) - Path("settled.json").write_text(settled.model_dump_json() if settled else "nothing") +@flow(agents=Agents, envs=Envs, params=FlowParams, name="flow") +async def run(task: str, *, agents: Agents, envs: Envs, params: FlowParams, + ctx: FlowContext) -> None: + human = agents["human"] + session = await human.spawn(env=envs["workspace"]) + settled = await human.run(task, session=session, output_schema=Settled) + Path("settled.json").write_text(settled.model_dump_json()) ''' @@ -2750,17 +2686,16 @@ async def test_the_person_asked_for_a_shape_is_asked_a_question_at_a_time( ) -> None: """A flow settles what only a person can settle, in the model it is going to run on. - The same road an agent's own question takes, so the interface shows each field as a + Whoever is outside the run is asked for it, so the interface shows each field as a question with what it will take under it, and the next line typed is the answer. """ monkeypatch.chdir(tmp_path) written(tmp_path, "flow", QUESTIONNAIRE) app = Humanize() async with app.run_test() as driver: - # As `/flow` sets it: the flow, what each of its agents runs -- which is nothing, - # since the only one it drives is the person -- and the places it asks for. - app._flow_named, app._models = "flow", [] - app._wanted = app._places_of("flow") + # As `/flow` sets it: the flow, and no agent role -- the only one it has is the + # person, and nobody chooses who that is. + set_up(app, "flow", {}) await driver.press(*"how should I do this") await driver.press("enter") diff --git a/tests/integration/tui/test_attach.py b/tests/integration/tui/test_attach.py index 74dd39d5..eda323df 100644 --- a/tests/integration/tui/test_attach.py +++ b/tests/integration/tui/test_attach.py @@ -24,7 +24,7 @@ from hmz.tui.monitor import short from hmz.tui.pick import Held, reads from tests.stubs import ShellAgent, ShellSession, written -from tests.tui.fixtures import transcript +from tests.tui.fixtures import holding, set_up, transcript from tests.tui.fixtures import until as waited if TYPE_CHECKING: @@ -61,18 +61,29 @@ HOLDING = """ import asyncio -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import LocalEnv, flow -@flow -def run(agents: tuple[AgentBase, AgentBase], task: str) -> None: - async def both() -> None: - # A turn apiece, open at the same time and neither answering until the fake CLI is - # let go: two agents working at once is the case tab is for. - await asyncio.gather(*(agent.aturn("hold") for agent in agents)) +class Agents(AgentCollection): + builder: Agent + reviewer: Agent - asyncio.run(both()) + +class Envs(EnvCollection): + workspace: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams, name="flow") +async def run(task: str, *, agents: Agents, envs: Envs, params: FlowParams, + ctx: FlowContext) -> None: + async def hold(agent: Agent) -> None: + session = await agent.spawn(env=envs["workspace"]) + await agent.run("hold", session=session) + + # A turn apiece, open at the same time and neither answering until the fake CLI is let + # go: two agents working at once is the case tab is for. + await asyncio.gather(hold(agents["builder"]), hold(agents["reviewer"])) """ @@ -149,15 +160,18 @@ async def _two_agents(app: Humanize, driver: Pilot[None], where: Path) -> None: driver: What is pumping it. where: The workspace it is running in. """ - app._flow_named = "flow" - app._models = [Runs("claude/m:high"), Runs("claude/m:high")] + set_up( + app, + "flow", + {"builder": Runs("claude/m:high"), "reviewer": Runs("claude/m:high")}, + ) await driver.press(*"do it") await driver.press("enter") await until(lambda: len(app._conversations()) == 2, driver) # Both working, which is what tab steps between: a turn that has not started is not one # to step onto, and one that has ended is not either. await until(lambda: len(app._working) == 2, driver) - assert app._agents # and it holds them until it is let go, so nothing here can race + assert app._run is not None # held until it is let go, so nothing here can race assert not (where / "go.txt").exists() @@ -186,8 +200,12 @@ async def _both_working( Each agent and the conversation it is working in, in the order the flow takes them. """ one, two = SteerableAgent(CONFIG), SteerableAgent(CONFIG) - app._agents = [one, two] - app._models = [Runs("claude/m:high"), Runs("codex/n:high")] + # Named for the roles they fill, as the run names the agent behind each session it opens. + one.rename("builder") + two.rename("reviewer") + holding(app, one, two) + app._models = {"builder": Runs("claude/m:high"), "reviewer": Runs("codex/n:high")} + app._declared = None # a flow nothing here loads, whose roles are these two first, second = one.new(), two.new() app._heard(one, first, Event(kind="begins", text="")) app._heard(two, second, Event(kind="begins", text="")) @@ -209,7 +227,7 @@ async def test_it_opens_on_the_transcript_every_agent_is_on(workspace: Path) -> assert app._attached == _EVERY assert app._reading() is None _let_go(workspace) - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None, driver) @pytest.mark.timeout(60) @@ -233,7 +251,7 @@ async def test_tab_steps_round_the_agents_that_are_working(workspace: Path) -> N assert app._attached == _EVERY _let_go(workspace) - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None, driver) @pytest.mark.timeout(60) @@ -253,7 +271,7 @@ async def test_shift_tab_steps_the_other_way_round(workspace: Path) -> None: assert app._attached == first _let_go(workspace) - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None, driver) @pytest.mark.timeout(60) @@ -311,8 +329,9 @@ async def test_every_conversation_of_one_agent_runs_down_the_same_transcript() - app = Humanize() async with app.run_test() as driver: agent = SteerableAgent(CONFIG) - app._agents = [agent] - app._models = [Runs("claude/m:high")] + agent.rename("builder") + holding(app, agent) + app._models = {"builder": Runs("claude/m:high")} first = agent.new() app._heard(agent, first, Event(kind="begins", text="")) app._heard(agent, first, Event(kind="text", text="the first round")) @@ -408,8 +427,9 @@ async def test_nothing_is_said_to_a_conversation_between_turns() -> None: app = Humanize() async with app.run_test() as driver: agent = SteerableAgent(CONFIG) - app._agents = [agent] - app._models = [Runs("claude/m:high")] + agent.rename("builder") + holding(app, agent) + app._models = {"builder": Runs("claude/m:high")} session = agent.new() # open, and no turn in it await driver.press(*"and this") @@ -459,7 +479,7 @@ async def test_a_flow_starting_reads_the_transcript_they_are_all_on( assert app._attached == _EVERY assert "that flow has gone" in transcript(app) _let_go(workspace) - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None, driver) @pytest.mark.timeout(60) @@ -509,10 +529,8 @@ def test_an_agent_holding_nothing_says_nothing_about_it() -> None: """Which is every agent of a flow that is not running, and how that line always read.""" runs = [Runs("claude/claude-opus-5:max")] - # What it may do is on the line whether or not anybody narrowed it: an agent nobody - # narrowed runs at what it was configured with, and the line says so rather than leaving - # the gap a reader could not tell from a setting that had gone missing. - at = "builder · claude/claude-opus-5:max · as configured" + # The role and what it runs, and nothing about what it may do: that is the flow's. + at = "builder · claude/claude-opus-5:max" assert reads(("builder",), runs) == [at] assert reads(("builder",), runs, [Held()]) == [at] @@ -594,8 +612,13 @@ async def test_the_diagram_marks_who_is_working_and_who_handed_to_whom() -> None app = Humanize() async with app.run_test() as driver: one, two = SteerableAgent(CONFIG), SteerableAgent(CONFIG) - app._agents = [one, two] - app._models = [Runs("claude/m:high"), Runs("codex/n:high")] + one.rename("builder") + two.rename("reviewer") + holding(app, one, two) + app._models = { + "builder": Runs("claude/m:high"), + "reviewer": Runs("codex/n:high"), + } first, second = one.new(), two.new() # A turn, and then a turn of the other agent: which is a handover between them. app._heard(one, first, Event(kind="begins", text="")) diff --git a/tests/integration/tui/test_board.py b/tests/integration/tui/test_board.py index e1865e14..b0aebe13 100644 --- a/tests/integration/tui/test_board.py +++ b/tests/integration/tui/test_board.py @@ -52,8 +52,12 @@ async def _opens(app: Humanize, driver: Pilot[None]) -> None: def _two(app: Humanize) -> tuple[AgentBase, AgentBase]: """Two agents of a flow, neither of which has taken a turn yet.""" one, two = SteerableAgent(CONFIG), SteerableAgent(CONFIG) + # Named for the roles they fill, as a run names the agent behind each session it opens. + one.rename("builder") + two.rename("reviewer") app._agents = [one, two] - app._models = [Runs("claude/m:high"), Runs("codex/n:high")] + app._models = {"builder": Runs("claude/m:high"), "reviewer": Runs("codex/n:high")} + app._declared = None # a flow nothing here loads, whose roles are these two return one, two diff --git a/tests/integration/tui/test_boxes.py b/tests/integration/tui/test_boxes.py index 7d4cfc25..dceb0164 100644 --- a/tests/integration/tui/test_boxes.py +++ b/tests/integration/tui/test_boxes.py @@ -19,7 +19,7 @@ import pytest from hmz.tui import Humanize -from hmz.tui.pick import Confirms, Leaves, Popup, Reports, Unbounded +from hmz.tui.pick import Confirms, Leaves, Popup, Reports from tests.tui.fixtures import until if TYPE_CHECKING: @@ -35,7 +35,6 @@ "opens", [ pytest.param(Confirms, id="confirms"), - pytest.param(Unbounded, id="unbounded"), pytest.param(partial(Leaves, held=False), id="leaves"), pytest.param(Reports, id="reports"), ], diff --git a/tests/integration/tui/test_btw.py b/tests/integration/tui/test_btw.py index 4ddab1db..0a6c944a 100644 --- a/tests/integration/tui/test_btw.py +++ b/tests/integration/tui/test_btw.py @@ -13,7 +13,7 @@ from hmz.tui import Humanize from hmz.tui.app import _COMMANDS from hmz.tui.btw import format_snapshot -from tests.tui.fixtures import transcript +from tests.tui.fixtures import holding, transcript if TYPE_CHECKING: import os @@ -69,9 +69,10 @@ async def test_btw_is_offered_and_does_not_enqueue_a_primary_message() -> None: """The command is a side turn, not another line for the running flow.""" app = Humanize() primary = MainAgent(CONFIG) + primary.rename("builder") held = primary.new() - app._agents = [primary] - app._models = [Runs("claude/m:high")] + holding(app, primary) + app._models = {"builder": Runs("claude/m:high")} app._queued = ["keep working"] app._given = [(primary.id, "already handed")] app._monitor.begins(primary.id, "m") diff --git a/tests/integration/tui/test_budget.py b/tests/integration/tui/test_budget.py index f6b4b959..fd724e5d 100644 --- a/tests/integration/tui/test_budget.py +++ b/tests/integration/tui/test_budget.py @@ -1,33 +1,26 @@ """What a run of a flow may spend, set from the menu that sets everything else up. -A row on the page the flow's agents are on, because it is a setting of the run rather than of -the flow: the flow declares at most a default and never holds itself to one. What is checked -is that the row says what the run is held to without being opened, that what is set there is -written down beside what the flow was set up with and read back, and that saving a run with -nothing at all to stop it asks whether that is what was meant -- except of a flow that says in -its own file that it is meant to run that way, which `chat` does. +A row on the page the flow's roles are on, because it is a setting of the run rather than of +the flow. What is checked is that the row says what the run is held to without being opened, +that what is set there is written down beside what the flow was set up with and read back, +and that a flow is not saved until a run of it is given one -- except `chat`, the flow +humanize ships, which is a conversation and stops when the person does. """ from __future__ import annotations +import datetime from typing import TYPE_CHECKING, cast import pytest from textual.widgets import Label, OptionList -from hmz.coganchor.agents import Allowance from hmz.coganchor.backends import Model +from hmz.flows import Budget from hmz.runtime.kept import Runs from hmz.runtime.settings import Settings from hmz.tui import Humanize -from hmz.tui.pick import ( - _BUDGET, - _SAVE, - Configures, - Flows, - Unbounded, - budget_of, -) +from hmz.tui.pick import _BUDGET, _SAVE, Configures, Flows, budget_of from tests.integration.tui.test_app import onto, opens, rows from tests.stubs import written from tests.tui.fixtures import until @@ -40,30 +33,25 @@ #: One backend, so that the menu has something to set an agent up as and can be saved. _INSTALLED = {"claude": (Model("m", ("high",)),)} -#: A flow with no opinion about what a run of it is worth, which is what most flows are. +#: A flow like most: one agent, and a run of it is given a budget. QUIET = """ -from hmz._legacy_flows import Agent, flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow -@flow -def run(agents: tuple[Agent], task: str) -> None: - pass -""" +class Agents(AgentCollection): + worker: Agent -#: And one that says it is meant to run under nothing at all, as `chat` does. -LOOSE = """ -from hmz._legacy_flows import Agent, Allowance, flow - -@flow(budget=Allowance()) -def run(agents: tuple[Agent], task: str) -> None: +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def quiet(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: pass """ @pytest.fixture def flows(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: - """Puts the two flows where this project's own would be, with a backend to run them.""" + """Puts the flow where this project's own would be, with a backend to run it.""" import hmz.tui.app import hmz.tui.pick @@ -72,13 +60,12 @@ def flows(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: where = tmp_path / ".humanize" / "flows" where.mkdir(parents=True) written(where, "quiet", QUIET) - written(where, "loose", LOOSE) return where async def _into(app: Humanize, driver: Pilot[None], flow: str) -> Flows: """Opens the menu already inside one flow, which is where the budget row is.""" - await driver.press(*f"/flow local/{flow}") + await driver.press(*f"/flow {flow}") await driver.press("enter") await until(lambda: isinstance(app.screen, Flows), driver) sheet = cast("Flows", app.screen) @@ -92,17 +79,22 @@ def _said(app: Humanize) -> str: return str(listing.get_option_at_index(rows(app).index(_BUDGET)).prompt) +def _under(app: Humanize) -> str: + """What the sheet on top says under its list.""" + return str(app.screen.query_one("#tuning", Label).content) + + @pytest.mark.timeout(60) async def test_the_row_says_what_the_run_is_held_to_without_being_opened( flows: Path, ) -> None: - """An allowance nobody can see without opening something is one nobody checks.""" + """A budget nobody can see without opening something is one nobody checks.""" app = Humanize() async with app.run_test() as driver: - await _into(app, driver, "quiet") + await _into(app, driver, "local/quiet") assert rows(app) == ["0", _BUDGET, _SAVE] - assert "nothing stops this run" in _said(app) + assert "none yet" in _said(app) @pytest.mark.timeout(60) @@ -112,23 +104,24 @@ async def test_what_is_set_there_is_kept_and_read_back( """Beside what the flow was set up with, since it is a setting of the run beside it.""" app = Humanize() async with app.run_test() as driver: - await _into(app, driver, "quiet") + await _into(app, driver, "local/quiet") await opens(app, driver, _BUDGET) await until(lambda: isinstance(app.screen, Configures), driver) sheet = cast("Configures", app.screen) - # Three rows and nothing else: hours, millions of output tokens, dollars. - assert rows(app) == ["hours", "tokens", "dollars"] + # The four `-b` takes, and nothing else. + assert rows(app) == ["duration", "cost", "output_tokens", "graceful"] - await driver.press("right") # hours: 0 -> 1 - await driver.press("down", "right") # tokens: 0 -> 1 + await driver.press(*"1h") # a duration is written, as `-b` writes one + await driver.press("down", "right") # cost: 0 -> 1 await driver.pause() - assert (sheet._typed_in["hours"], sheet._typed_in["tokens"]) == ("1.0", "1.0") + assert (sheet._typed_in["duration"], sheet._typed_in["cost"]) == ("1h", "1.0") await driver.press("enter") await until(lambda: isinstance(app.screen, Flows), driver) # Said on the row it came back to, so that it is read rather than remembered. - assert "stops at 1h, 1M out" in _said(app) + assert "stops at 1h" in _said(app) + assert "$1.00" in _said(app) await onto(app, driver, _SAVE) await driver.press("enter") @@ -136,169 +129,68 @@ async def test_what_is_set_there_is_kept_and_read_back( # Written down under the flow, beside its agents, and read back by the next interface. assert Settings(tmp_path).budget("local/quiet") == { - "hours": 1.0, - "tokens": 1.0, - "dollars": 0.0, + "duration": "PT1H", + "cost": 1.0, + "output_tokens": None, + "graceful": True, } again = Humanize() - assert again._budget == Allowance(hours=1.0, tokens=1.0) + assert again._budget == Budget(duration=datetime.timedelta(hours=1), cost=1) @pytest.mark.timeout(60) -async def test_saving_a_run_nothing_will_stop_asks_whether_that_is_what_was_meant( +async def test_a_flow_is_not_saved_until_a_run_of_it_has_a_budget( flows: Path, tmp_path: Path ) -> None: - """Three caps and none of them set is a run that goes until somebody notices. - - A fair thing to ask for and a poor thing to arrive at by not answering three questions, - and the two look identical afterwards -- so it is asked once, where it can be changed. - """ + """A run is given one before anything runs, and a menu saved without is a run refused.""" app = Humanize() async with app.run_test() as driver: - await _into(app, driver, "quiet") + sheet = await _into(app, driver, "local/quiet") await onto(app, driver, _SAVE) await driver.press("enter") - await until(lambda: isinstance(app.screen, Unbounded), driver) + await driver.pause() - # And the second answer goes back to the menu holding everything it was holding, - # which is where a budget is set. - await driver.press("down", "enter") - await until(lambda: isinstance(app.screen, Flows), driver) + assert app.screen is sheet # still here, holding everything it was holding + assert "given a budget" in _under(app) assert Settings(tmp_path).flow != "local/quiet" - # Asked again on the way past, and meant this time. - await onto(app, driver, _SAVE) - await driver.press("enter") - await until(lambda: isinstance(app.screen, Unbounded), driver) - await driver.press("enter") - await until(lambda: not isinstance(app.screen, Flows), driver) - - assert Settings(tmp_path).flow == "local/quiet" - - -@pytest.mark.timeout(60) -async def test_a_run_with_a_cap_on_it_is_not_asked_about( - flows: Path, tmp_path: Path -) -> None: - """The question is about a run nothing will stop, and one cap is something.""" - Settings(tmp_path).remember( - "local/quiet", ("",), [Runs("claude/m:high")], budget={"hours": 2} - ) - app = Humanize() - async with app.run_test() as driver: - sheet = await _into(app, driver, "quiet") - assert "stops at 2h" in _said(app) - - await onto(app, driver, _SAVE) - await driver.press("enter") - await until(lambda: app.screen is not sheet, driver) - - assert not isinstance(app.screen, Unbounded) - - -@pytest.mark.timeout(60) -async def test_saving_a_cap_nothing_can_price_asks_the_same_question( - flows: Path, tmp_path: Path -) -> None: - """Fifty dollars on a model nobody prices is a run with no limit on it at all. - - Which is the run a benchmark of sixteen cells actually started: capped in the settings, - and with nothing on the machine that can read the cap. So it reaches the same box the run - capped on nothing reaches, and the box says which cap and why rather than the three that - were not set -- the wording being the one `hmz exec` prints on its way past. - """ - Settings(tmp_path).remember( - "local/quiet", ("",), [Runs("claude/m:high")], budget={"dollars": 50} - ) - app = Humanize() - async with app.run_test() as driver: - await _into(app, driver, "quiet") - assert "stops at $50" in _said(app) - - await onto(app, driver, _SAVE) - await driver.press("enter") - await until(lambda: isinstance(app.screen, Unbounded), driver) - - # Read off the label rather than off the attribute: the whole of this is that a - # person sees it, and a line that is set and not drawn is the log line again. - shown = app.screen.query_one("#about", Label) - assert shown.display - assert "nothing here can read dollars" in str(shown.content).lower() - - -@pytest.mark.timeout(60) -async def test_a_token_cap_on_a_cli_that_counts_nothing_is_asked_about_too( - flows: Path, tmp_path: Path -) -> None: - """A CLI somebody added by hand is driven over a protocol that counts nothing at all. - - So ten million output tokens on one of those is a cap that will never bite either, and the - menu asks about it exactly as `hmz exec` says it -- which is the whole of this: the box - and the line are one question about one run, and a menu that could only see the money - would be the same split again a dimension over. - """ - from hmz.coganchor import backends - - backends.remember("spoken", ["spoken"]) - Settings(tmp_path).remember( - "local/quiet", ("",), [Runs("spoken/m:high")], budget={"tokens": 10} - ) - app = Humanize() - async with app.run_test() as pilot: - await _into(app, pilot, "quiet") - - await onto(app, pilot, _SAVE) - await pilot.press("enter") - await until(lambda: isinstance(app.screen, Unbounded), pilot) - - shown = app.screen.query_one("#about", Label) - assert "nothing here can read tokens" in str(shown.content).lower() - @pytest.mark.timeout(60) -async def test_a_cap_the_run_can_read_is_not_asked_about( - flows: Path, tmp_path: Path, priced: str +async def test_a_budget_that_limits_nothing_is_refused_where_it_is_typed( + flows: Path, ) -> None: - """The control: the same allowance on a model somebody lists is a cap that will bite.""" - Settings(tmp_path).remember( - "local/quiet", ("",), [Runs(f"claude/{priced}:high")], budget={"dollars": 50} - ) + """At least one limit, which is what makes it a budget: none is refused on the sheet.""" app = Humanize() async with app.run_test() as driver: - sheet = await _into(app, driver, "quiet") + await _into(app, driver, "local/quiet") + await opens(app, driver, _BUDGET) + await until(lambda: isinstance(app.screen, Configures), driver) - await onto(app, driver, _SAVE) await driver.press("enter") - await until(lambda: app.screen is not sheet, driver) - - assert not isinstance(app.screen, Unbounded) + await driver.pause() - assert Settings(tmp_path).flow == "local/quiet" + assert isinstance(app.screen, Configures) + assert "at least one" in _under(app) @pytest.mark.timeout(60) -async def test_a_clock_beside_a_cap_nothing_can_price_is_not_asked_about( - flows: Path, tmp_path: Path -) -> None: - """The money cannot be read and the hours can, so something still stops the run. - - Which is the case the benchmark survived on, and it must go on surviving on it. - """ +async def test_a_run_with_a_budget_saves_at_once(flows: Path, tmp_path: Path) -> None: + """What was set is read back as the menu opens, and saving it asks nothing.""" Settings(tmp_path).remember( "local/quiet", - ("",), - [Runs("claude/m:high")], - budget={"hours": 0.2, "dollars": 1}, + {"worker": Runs("claude/m:high")}, + budget={"duration": "PT2H"}, ) app = Humanize() async with app.run_test() as driver: - sheet = await _into(app, driver, "quiet") + sheet = await _into(app, driver, "local/quiet") + assert "stops at 2h" in _said(app) await onto(app, driver, _SAVE) await driver.press("enter") await until(lambda: app.screen is not sheet, driver) - assert not isinstance(app.screen, Unbounded) + assert Settings(tmp_path).flow == "local/quiet" @pytest.mark.timeout(60) @@ -307,75 +199,50 @@ async def test_a_flow_run_without_the_menu_is_still_held_to_what_was_set( ) -> None: """`$flow ` runs a flow this workspace has set up without opening the menu. - Which is the whole point of that line -- and a path that dropped the allowance on the way - would start an unbounded run out of a workspace whose settings say six hours, with nothing - asked either, the question living on the menu that did not open. + Which is the whole point of that line -- and a path that dropped the budget on the way + would start a run the runtime refuses, out of a workspace whose settings say six hours. """ Settings(tmp_path).remember( - "local/quiet", ("",), [Runs("claude/m:high")], budget={"hours": 6} + "local/quiet", + {"worker": Runs("claude/m:high")}, + budget={"duration": "PT6H"}, ) app = Humanize() async with app.run_test(): held = app._remembered_for("local/quiet") assert held is not None - assert held.budget == Allowance(hours=6) + assert held.budget == Budget(duration=datetime.timedelta(hours=6)) @pytest.mark.timeout(60) -async def test_setting_every_dimension_back_to_nothing_forgets_it( +async def test_one_remembered_with_no_budget_is_asked_about_rather_than_run( flows: Path, tmp_path: Path ) -> None: - """Rather than writing three zeros down, which would override the flow for good. - - A flow is back under what it says for itself by there being nothing remembered for it, so - an allowance that caps nothing has to be written down as nothing. - """ - Settings(tmp_path).remember( - "local/quiet", ("",), [Runs("claude/m:high")], budget={"hours": 6} - ) + """A flow set up before it had one is set up again, rather than started to be refused.""" + Settings(tmp_path).remember("local/quiet", {"worker": Runs("claude/m:high")}) app = Humanize() - async with app.run_test() as driver: - await _into(app, driver, "quiet") - assert "stops at 6h" in _said(app) - - await opens(app, driver, _BUDGET) - await until(lambda: isinstance(app.screen, Configures), driver) - await driver.press("left") # hours: 6 -> 5 - for _ in range(5): - await driver.press("left") # and down to nothing - await driver.pause() - await driver.press("enter") - await until(lambda: isinstance(app.screen, Flows), driver) - await onto(app, driver, _SAVE) - await driver.press("enter") - await until(lambda: isinstance(app.screen, Unbounded), driver) - await driver.press("enter") - await until(lambda: not isinstance(app.screen, Flows), driver) - - assert Settings(tmp_path).budget("local/quiet") == {} - assert budget_of("local/quiet") is None + async with app.run_test(): + assert app._remembered_for("local/quiet") is None + assert budget_of("local/quiet") is None @pytest.mark.timeout(60) -async def test_a_flow_that_says_it_runs_unbounded_is_never_asked( - flows: Path, tmp_path: Path +async def test_the_conversation_humanize_ships_is_never_asked_for_one( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: - """`@flow(budget=Allowance())` is a flow claiming an unbounded run is what it is for. + """`chat` stops when the person does, and runs under no budget of anybody's.""" + import hmz.tui.app - Which is the whole of the exemption and the whole of why `chat` is not named in the menu, - the command line and the settings: it is one line in the flow's own file, and any flow may - make the same claim. - """ + monkeypatch.setattr(hmz.tui.app, "installed", lambda: dict(_INSTALLED)) app = Humanize() async with app.run_test() as driver: - sheet = await _into(app, driver, "loose") - assert "nothing stops this run" in _said(app) + sheet = await _into(app, driver, "chat") + assert "none needed" in _said(app) await onto(app, driver, _SAVE) await driver.press("enter") await until(lambda: app.screen is not sheet, driver) - assert not isinstance(app.screen, Unbounded) - - assert Settings(tmp_path).flow == "local/loose" + assert Settings(tmp_path).flow == "chat" + assert app._budget is None diff --git a/tests/integration/tui/test_config.py b/tests/integration/tui/test_config.py index bf217d4d..cb91bb7d 100644 --- a/tests/integration/tui/test_config.py +++ b/tests/integration/tui/test_config.py @@ -1,9 +1,9 @@ """Setting a flow up: the sheet between choosing the flow and choosing what runs it. -A flow says what it can be set up with by declaring a model, and this is that model with a -cursor on it. Nothing here knows what any of the settings mean: the types say how a value is -moved and the model says which combinations it will not take, so what is checked is that both -of those reach the person setting it up. +A flow says what it can be set up with by declaring its params, a `FlowParams` model, and this +is that model with a cursor on it. Nothing here knows what any of the settings mean: the types +say how a value is moved and the model says which combinations it will not take, so what is +checked is that both of those reach the person setting it up. """ from __future__ import annotations @@ -34,16 +34,19 @@ FLOW = ''' from typing import Literal -from pydantic import BaseModel, Field, model_validator +from pydantic import Field, model_validator -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow FIRST = {"section": "first · how loudly"} SECOND = {"section": "second · how far"} -class Config(BaseModel): +class Agents(AgentCollection): + worker: Agent + + +class Config(FlowParams): """What this flow takes.""" loud: bool = Field( @@ -66,51 +69,68 @@ def _settles(self) -> "Config": return self -@flow -def run(agents: tuple[AgentBase], task: str, config: Config | None = None) -> None: +@flow(agents=Agents, envs=EnvCollection, params=Config) +async def settable(task: str, *, agents: Agents, envs: EnvCollection, params: Config, + ctx: FlowContext) -> None: pass ''' #: A flow that takes settings but groups none of them, which is a sheet of one list. UNGROUPED = ''' -from pydantic import BaseModel, Field +from pydantic import Field + +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow + -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +class Agents(AgentCollection): + worker: Agent -class Config(BaseModel): +class Config(FlowParams): """What this flow takes.""" loud: bool = Field(default=False, description="say it twice") rounds: int = Field(default=3, description="how many times round") -@flow -def run(agents: tuple[AgentBase], task: str, config: Config | None = None) -> None: +@flow(agents=Agents, envs=EnvCollection, params=Config) +async def ungrouped(task: str, *, agents: Agents, envs: EnvCollection, params: Config, + ctx: FlowContext) -> None: pass ''' #: A flow that takes no setting up at all, which is what most of them are. PLAIN = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow -@flow -def run(agents: tuple[AgentBase], task: str) -> None: +class Agents(AgentCollection): + worker: Agent + + +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def plain(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: pass """ @pytest.fixture def flows(tmp_path: Path) -> Path: - """Puts both flows where this project's own would be.""" + """Puts the flows where this project's own would be, each with a budget set for a run. + + Set rather than asked, since saving a flow a run of which has none is refused, and none + of these tests is about that: the workspace is left set up on `chat`, as it opens. + """ where = tmp_path / ".humanize" / "flows" where.mkdir(parents=True) written(where, "settable", FLOW) written(where, "plain", PLAIN) written(where, "ungrouped", UNGROUPED) + kept = Settings(tmp_path) + for one in ("settable", "plain", "ungrouped"): + kept.remember(f"local/{one}", {}, budget={"cost": 1.0}) + kept.remember("chat", {}) return where @@ -324,13 +344,13 @@ async def test_how_it_was_set_up_is_kept_and_read_back( await driver.press("enter") await until(lambda: isinstance(app.screen, Flows), driver) await keeps(app, driver) - await until(lambda: app._config is not None, driver) + await until(lambda: app._params is not None, driver) - assert Settings(tmp_path).config("local/settable")["loud"] is True + assert Settings(tmp_path).params("local/settable")["loud"] is True # And a second interface opens on it, rather than back at the flow's own defaults. again = Humanize() - assert again._config is not None - assert again._config.model_dump()["loud"] is True + assert again._params is not None + assert again._params.model_dump()["loud"] is True @pytest.mark.timeout(60) @@ -389,14 +409,16 @@ def test_a_config_that_no_longer_fits_the_flow_is_started_over_from( ) -> None: """A settings file is a convenience, and one that has gone stale is not a reason to fail.""" Settings(tmp_path).remember( - "settable", ("",), [Runs("claude/opus:high")], {"gone": "away", "rounds": 99} + "local/settable", + {"worker": Runs("claude/opus:high")}, + params={"gone": "away", "rounds": 99}, ) app = Humanize() - from hmz.tui.pick import config_of + from hmz.tui.pick import params_of - assert config_of("settable", app.settings.config("settable")) is None + assert params_of("local/settable", app.settings.params("local/settable")) is None @pytest.mark.timeout(60) @@ -504,8 +526,8 @@ async def test_walking_past_how_the_flow_is_set_up_leaves_it_alone(flows: Path) await keeps(app, driver) await until(lambda: not isinstance(app.screen, Flows), driver) - assert app._config is not None - assert app._config.model_dump()["loud"] is True + assert app._params is not None + assert app._params.model_dump()["loud"] is True @pytest.mark.timeout(60) diff --git a/tests/integration/tui/test_epics.py b/tests/integration/tui/test_epics.py index f26eb11c..a609b09f 100644 --- a/tests/integration/tui/test_epics.py +++ b/tests/integration/tui/test_epics.py @@ -23,26 +23,31 @@ from hmz.tui.pick import Does, Epics from tests.integration.tui.test_app import onto, rows from tests.stubs import written -from tests.tui.fixtures import until +from tests.tui.fixtures import holding, until if TYPE_CHECKING: from pathlib import Path from textual.pilot import Pilot -#: A flow that says it can be picked up, and counts the runs of itself in what it is handed. +#: A flow that says it can be picked up, and counts the runs of itself in what it keeps. COUNTS = '''"""Counts the runs of itself.""" from pathlib import Path -from typing import Any -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow -@flow(resumable=True) -def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None: - state["rounds"] = state.get("rounds", 0) + 1 +class Agents(AgentCollection): + worker: Agent + + +@flow(agents=Agents, envs=EnvCollection, params=FlowParams, resumable=True) +async def counts(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: + state = ctx.state + assert state is not None + state["rounds"] = (state["rounds"] if "rounds" in state else 0) + 1 Path("rounds.txt").write_text(str(state["rounds"])) ''' @@ -51,25 +56,39 @@ def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None: from pathlib import Path -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow + + +class Agents(AgentCollection): + worker: Agent -@flow -def run(agents: tuple[AgentBase], task: str) -> None: +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def plain(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: Path("plain.txt").write_text(task) ''' #: One that opens a session and says one thing, so that there is a run with a trace in it. SPEAKS = '''"""Takes one turn, and says nothing about being picked up.""" -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import LocalEnv, flow + + +class Agents(AgentCollection): + worker: Agent + + +class Envs(EnvCollection): + workspace: LocalEnv -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - agents[0].new()(task) +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def speaks(task: str, *, agents: Agents, envs: Envs, params: FlowParams, + ctx: FlowContext) -> None: + worker = agents["worker"] + await worker.run(task, session=await worker.spawn(env=envs["workspace"])) ''' #: A `claude` that answers whatever it is told, since what is being tested is the run rather @@ -132,15 +151,12 @@ def workspace(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: def _ran(flow: str, task: str) -> None: """Runs one flow here, the way a command line would, so there is an epic to look at. - On the fake `claude` this suite puts on PATH rather than on a stand-in of our own: what a - run is picked up as is a command line naming what each agent runs, so the agents of a run - have to be agents something can name. + On the fake `claude` this suite puts on PATH: what a run is picked up as is the specs + each of its roles was given, so the agents of a run have to be agents something can name. """ - from hmz.coganchor.agents import driver - from hmz.runtime.runner import Runner + from hmz.runtime import Hmz - agent, config = driver("claude") - Runner(flow, [agent(config(model="m", effort="high"))]).run(task) + Hmz().run(flow, task, agents={"worker": "claude/m:high"}, budget={"cost": 1}).run() async def _open(app: Humanize, driver: Pilot[None]) -> Epics: @@ -353,7 +369,7 @@ def test_the_trace_from_the_menu_is_of_that_run_and_of_nothing_else( from hmz.tui.pick import exported del workspace - epic = Epic("plain", [], "go") + epic = Epic("plain", "go") epic.write("opened", agent="actor", backend="claude", session="one") epic.write("opened", agent="reviewer", backend="claude", session="two") collect = unittest.mock.Mock(return_value={"otherData": {}}) @@ -382,8 +398,8 @@ async def test_a_run_of_a_flow_marked_since_can_be_picked_up_too( """What can be done with a run is what its flow says now, not what the run recorded. A flow is a directory on disk: one marked resumable after a run of it is one whose older - runs can be carried on, and the run's own record is what it was rather than what there is - to do with it today. + runs are offered to be carried on -- and that one kept no journal, so the list does not + mark it as one that can be, and carrying it on says why (see the test after this). """ _ran("plain", "go") (epic,) = epics(workspace) @@ -399,7 +415,7 @@ async def test_a_run_of_a_flow_marked_since_can_be_picked_up_too( app = Humanize() async with app.run_test() as driver: await _open(app, driver) - assert "can be picked up" in str( + assert "can be picked up" not in str( app.screen.query_one("#choices", OptionList).get_option_at_index(0).prompt ) @@ -457,7 +473,7 @@ async def test_carrying_one_on_is_refused_while_a_flow_runs_and_not_after( app = Humanize() async with app.run_test() as driver: - app._agents = [ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high"))] + holding(app, ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high"))) sheet = await _open(app, driver) await driver.press("enter") await until(lambda: isinstance(app.screen, Does), driver) @@ -470,7 +486,7 @@ async def test_carrying_one_on_is_refused_while_a_flow_runs_and_not_after( # And the same list, once the flow is over, picks the run up rather than repeating # itself about a run that has gone. - app._agents = [] + app._run = None await driver.press("enter") await until(lambda: isinstance(app.screen, Does), driver) await onto(app, driver, "resume") diff --git a/tests/integration/tui/test_export.py b/tests/integration/tui/test_export.py index e26bf4e8..a21bf074 100644 --- a/tests/integration/tui/test_export.py +++ b/tests/integration/tui/test_export.py @@ -36,13 +36,23 @@ #: A flow that opens one session and says one thing, so that there is a run to package up. PLAIN = '''"""Runs once, and says what it was told.""" -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import LocalEnv, flow -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - agents[0].new()(task) +class Agents(AgentCollection): + worker: Agent + + +class Envs(EnvCollection): + workspace: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def plain(task: str, *, agents: Agents, envs: Envs, params: FlowParams, + ctx: FlowContext) -> None: + worker = agents["worker"] + await worker.run(task, session=await worker.spawn(env=envs["workspace"])) ''' #: A `claude` that answers whatever it is told and logs the session where Claude Code logs @@ -82,11 +92,11 @@ def workspace(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: def _ran(task: str) -> None: """Runs the flow here the way a command line would, so there is an epic to export.""" - from hmz.coganchor.agents import driver - from hmz.runtime.runner import Runner + from hmz.runtime import Hmz - agent, config = driver("claude") - Runner("plain", [agent(config(model="m", effort="high"))]).run(task) + Hmz().run( + "plain", task, agents={"worker": "claude/m:high"}, budget={"cost": 1} + ).run() def _held(at: Path) -> dict[str, str]: diff --git a/tests/integration/tui/test_flowverses.py b/tests/integration/tui/test_flowverses.py index 66538568..2ebc67cf 100644 --- a/tests/integration/tui/test_flowverses.py +++ b/tests/integration/tui/test_flowverses.py @@ -38,14 +38,24 @@ #: A flow, as short as one can be: what is being fetched is the file, not what it does. FLOW = '''"""Somebody else's loop, fetched from somewhere else.""" -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import LocalEnv, flow -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent.new()(task) +class Agents(AgentCollection): + worker: Agent + + +class Envs(EnvCollection): + workspace: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def run(task: str, *, agents: Agents, envs: Envs, params: FlowParams, + ctx: FlowContext) -> None: + """Somebody else's loop.""" + worker = agents["worker"] + await worker.run(task, session=await worker.spawn(env=envs["workspace"])) ''' @@ -227,7 +237,7 @@ async def test_a_flow_says_what_it_does_beside_its_name() -> None: .prompt ) - assert "one agent, one session" in drawn + assert "Talks to one agent" in drawn @pytest.mark.timeout(60) @@ -353,42 +363,39 @@ async def test_the_flow_that_is_picked_is_the_one_that_was_chosen(theirs: Path) #: A file that is three flows and no `run`, which is what humanize1 is. -THREE = '''"""Three phases of one thing, which are three things to run.""" - -from typing import NamedTuple - -from pydantic import BaseModel +THREE = '''"""Phases of one thing, which are things to run apiece.""" -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow -class Drafting(NamedTuple): +class Drafting(AgentCollection): """The one that writes.""" - drafter: AgentBase + drafter: Agent -class Building(NamedTuple): +class Building(AgentCollection): """The one that builds, and the one that reads it.""" - builder: AgentBase - reviewer: AgentBase + builder: Agent + reviewer: Agent -class Wide(BaseModel): +class Wide(FlowParams): """What the first phase takes.""" n: int = 6 -@flow(name="gen-idea") -def gen_idea(agents: Drafting, task: str, config: Wide | None = None) -> None: +@flow(agents=Drafting, envs=EnvCollection, params=Wide, name="gen-idea") +async def gen_idea(task: str, *, agents: Drafting, envs: EnvCollection, params: Wide, + ctx: FlowContext) -> None: """Opens a loose idea into a draft.""" -@flow(name="rlcr") -def rlcr(agents: Building, task: str) -> None: +@flow(agents=Building, envs=EnvCollection, params=FlowParams, name="rlcr") +async def rlcr(task: str, *, agents: Building, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: """Builds it, under review.""" ''' diff --git a/tests/integration/tui/test_flowverses_menu.py b/tests/integration/tui/test_flowverses_menu.py index d9fb3a01..a8f376ae 100644 --- a/tests/integration/tui/test_flowverses_menu.py +++ b/tests/integration/tui/test_flowverses_menu.py @@ -36,14 +36,24 @@ #: A flow, as short as one can be: what is being fetched is the file, not what it does. FLOW = '''"""Somebody else's loop, fetched from somewhere else.""" -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import LocalEnv, flow -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent.new()(task) +class Agents(AgentCollection): + worker: Agent + + +class Envs(EnvCollection): + workspace: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def run(task: str, *, agents: Agents, envs: Envs, params: FlowParams, + ctx: FlowContext) -> None: + """Somebody else's loop, fetched from somewhere else.""" + worker = agents["worker"] + await worker.run(task, session=await worker.spawn(env=envs["workspace"])) ''' diff --git a/tests/integration/tui/test_keys.py b/tests/integration/tui/test_keys.py index 96a66b89..53662e8b 100644 --- a/tests/integration/tui/test_keys.py +++ b/tests/integration/tui/test_keys.py @@ -47,7 +47,6 @@ Retries, Sheet, Speaks, - Unbounded, ) from tests.integration.tui.test_app import into_agent, into_flows, onto, rows from tests.tui.fixtures import until @@ -105,7 +104,6 @@ def test_the_keys_are_written_in_one_place() -> None: "opens", [ pytest.param(Confirms, id="confirms"), - pytest.param(Unbounded, id="unbounded"), pytest.param(partial(Leaves, held=True), id="leaves"), pytest.param(Reports, id="reports"), pytest.param(Fetches, id="fetches"), diff --git a/tests/integration/tui/test_leaving.py b/tests/integration/tui/test_leaving.py index 2c325e61..f88bc041 100644 --- a/tests/integration/tui/test_leaving.py +++ b/tests/integration/tui/test_leaving.py @@ -15,11 +15,10 @@ import pytest -from hmz.runtime.kept import Runs from hmz.tui import Humanize from hmz.tui.pick import DETACHES, STAYS, STOPS, Leaves from tests.stubs import written -from tests.tui.fixtures import transcript, until +from tests.tui.fixtures import ONE, set_up, transcript, until if TYPE_CHECKING: from pathlib import Path @@ -28,18 +27,7 @@ #: A flow whose one turn does not end until something else is said to it, so that a test can #: ask what happens to a run that is still running. -FLOW = """ -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - session = agents[0].new() - Path("said.txt").write_text(session(task) + "\\n") -""" +FLOW = ONE #: A `claude` that does not answer until it is told something else, so that a turn stays open. PATIENT = """ @@ -130,12 +118,9 @@ async def test_the_key_that_leaves_asks_what_leaving_asks(workspace: Path) -> No written(workspace, "flow", FLOW) app = Humanize(session=Holding()) async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await _says(app, driver, "start") - await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), - driver, - ) + await until(lambda: any(agent.sessions for agent in app._agents), driver) await driver.press("ctrl+q") await until(lambda: isinstance(app.screen, Leaves), driver) @@ -158,12 +143,9 @@ async def test_exit_with_nothing_running_asks_nothing() -> None: async def _asks(app: Humanize, driver: Pilot[None]) -> None: """Starts the flow, waits for its turn to be open, and asks the interface to close.""" - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await _says(app, driver, "start") - await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), - driver, - ) + await until(lambda: any(agent.sessions for agent in app._agents), driver) await _says(app, driver, "/exit") await until(lambda: isinstance(app.screen, Leaves), driver) @@ -214,7 +196,7 @@ async def test_leaving_it_running_lets_go_of_the_terminal_instead( assert holding.let_go == 1 assert app.is_running # the flow is still going, where nothing is reading it - assert app._agents + assert app._run is not None @pytest.mark.timeout(120) diff --git a/tests/integration/tui/test_models.py b/tests/integration/tui/test_models.py index 07f46074..7dc041b9 100644 --- a/tests/integration/tui/test_models.py +++ b/tests/integration/tui/test_models.py @@ -17,9 +17,10 @@ from hmz.coganchor.backends import Model from hmz.runtime.kept import Runs +from hmz.runtime.settings import Settings from hmz.tui import Humanize from hmz.tui.pick import Agent, Catalogue, Clis -from tests.integration.tui.test_app import into_agent, keeps, onto, opens, rows +from tests.integration.tui.test_app import into_agent, keeps, opens, rows from tests.stubs import written from tests.tui.fixtures import until @@ -45,46 +46,40 @@ HERE = ''' """One agent, working where the flow is.""" -from typing import NamedTuple - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow -from tests.stubs import written +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import LocalEnv, flow -class Agents(NamedTuple): +class Agents(AgentCollection): """Just the one.""" - builder: AgentBase + builder: Agent + +class Envs(EnvCollection): + workspace: LocalEnv -@flow -def run(agents: Agents, task: str) -> None: + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def here(task: str, *, agents: Agents, envs: Envs, params: FlowParams, + ctx: FlowContext) -> None: pass ''' -GOALS_OFF = ( - HERE.replace( - "from typing import NamedTuple", "from typing import Annotated, NamedTuple" - ) - .replace( - "from hmz.coganchor.agents import AgentBase", - "from hmz.coganchor.agents import AgentBase, AgentDefaults", - ) - .replace( - "builder: AgentBase", - "builder: Annotated[AgentBase, AgentDefaults(goals=False)]", - ) -) - @pytest.fixture def flows(tmp_path: Path) -> Path: - """Puts the flow where this project's own would be.""" + """Puts the flow where this project's own would be, with a budget set for a run of it. + + Set rather than asked, since saving a flow a run of which has none is refused, and none + of these tests is about that: the workspace is left set up on `chat`, as it opens. + """ where = tmp_path / ".humanize" / "flows" where.mkdir(parents=True) written(where, "here", HERE) - written(where, "goals_off", GOALS_OFF) + kept = Settings(tmp_path) + kept.remember("here", {}, budget={"cost": 1.0}) + kept.remember("chat", {}) return where @@ -127,65 +122,6 @@ def _rows(app: Humanize) -> int: return len(app.screen.query_one("#choices", OptionList).options) -@pytest.mark.timeout(60) -@unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) -async def test_what_the_flow_said_about_goals_survives_the_sheet_untouched( - _installed: unittest.mock.MagicMock, # noqa: PT019 - flows: Path, -) -> None: - """It is not a row, so the sheet carries it across rather than answering it. - - Whether goals are available is the flow's to say, and a sheet that reset it to a default - because it never showed it would be a sheet quietly overruling the flow. - """ - app = Humanize() - async with app.run_test() as driver: - await _to_the_agent(app, driver, "goals_off") - assert "goals" not in rows(app) - - await onto(app, driver, "effort") - await driver.press("right") - await driver.pause() - - await keeps(app, driver) - await keeps(app, driver) - - assert app._models[0].goals is False - assert app.settings.agents("goals_off")[0].goals is False - - -@pytest.mark.timeout(60) -@unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) -async def test_opening_directly_uses_the_agent_place_goal_suggestion( - _installed: unittest.mock.MagicMock, # noqa: PT019 - flows: Path, -) -> None: - app = Humanize(flow="goals_off") - - async with app.run_test() as driver: - await driver.pause() - - assert app._models == [Runs("claude/claude-nine:high", goals=False)] - - -def test_a_goal_choice_is_written_to_the_agent_config( - flows: Path, -) -> None: - from hmz.coganchor.agents import ClaudeCodeAgent, ClaudeCodeAgentConfig - - app = Humanize( - flow="goals_off", - agents=[Runs("claude/claude-nine:high", goals=False)], - ) - made = ClaudeCodeAgent(ClaudeCodeAgentConfig(model="claude-nine", effort="high")) - - (configured,) = app._as_they_were_set_up([made]) - - assert configured is not made - assert configured.config.goals is False - assert not configured.goals_enabled - - @pytest.mark.timeout(60) @unittest.mock.patch("hmz.tui.app.installed", return_value=UNASKED) async def test_a_cli_that_has_not_said_what_it_runs_says_which_key_asks_it( @@ -253,7 +189,7 @@ async def test_a_name_in_the_catalogue_is_the_one_a_turn_runs_at( await keeps(app, driver) await keeps(app, driver) - assert app._models == [Runs("claude/fable:high")] + assert app._models == {"builder": Runs("claude/fable:high")} @pytest.mark.timeout(60) @@ -362,10 +298,13 @@ def note(cli: str, provider: str = "", seconds: float = 0.0) -> tuple[Model, ... monkeypatch.setattr(hmz.coganchor.models, "ask", note) app = Humanize() - assert app._models == [] + assert app._models == {} async with app.run_test() as driver: - await until(lambda: app._models == [Runs("claude/claude-nine:high")], driver) + await until( + lambda: app._models == {"assistant": Runs("claude/claude-nine:high")}, + driver, + ) await until(lambda: asked == ["claude", "codex", "dsh"], driver) assert asked[0] == "claude" diff --git a/tests/integration/tui/test_providers.py b/tests/integration/tui/test_providers.py index 83611382..ba4da4c6 100644 --- a/tests/integration/tui/test_providers.py +++ b/tests/integration/tui/test_providers.py @@ -43,7 +43,7 @@ opens, rows, ) -from tests.tui.fixtures import transcript, until +from tests.tui.fixtures import set_up, transcript, until if TYPE_CHECKING: from pathlib import Path @@ -419,13 +419,13 @@ async def test_the_account_an_agent_runs_as_is_the_first_thing_asked_about_it( # And on the line above the prompt, beside what it runs. assert "deepseek" in str(app.query_one("#above", Static).content) - chosen = Runs("claude/claude-opus-5:high", "", "", "deepseek") - assert app._models == [chosen] - assert app.settings.agents(app._flow_named) == [chosen] + chosen = Runs("claude/claude-opus-5:high", "deepseek") + assert app._models == {"assistant": chosen} + assert app.settings.agents(app._flow_named) == {"assistant": chosen} # And what it may do between the two: nobody narrowed this one, so the line says what # that comes to rather than leaving a gap where a rung would be. assert reads(("builder",), [chosen]) == [ - "builder · claude/claude-opus-5:high · as configured · deepseek" + "builder · claude/claude-opus-5:high · deepseek" ] @@ -488,7 +488,7 @@ def note(cli: str, provider: str = "", seconds: float = 0.0) -> tuple[Model, ... made = providers.find("claude", "mine") assert made is not None assert dict(made.env) == {"ANTHROPIC_API_KEY": "not-a-key"} - assert app._models == [Runs("claude/claude-opus-5:high", "", "", "mine")] + assert app._models == {"assistant": Runs("claude/claude-opus-5:high", "mine")} # The backends installed here are asked as the interface opens, and an account as it # lands: an account is made in order to run turns as, and which models those turns may # name is the account's rather than this machine's. @@ -526,7 +526,7 @@ async def test_making_one_and_walking_out_of_it_changes_nothing( app = Humanize() was = None async with app.run_test() as driver: - was = list(app._models) + was = dict(app._models) await into_flows(app, driver) await into_agent(app, driver) await opens(app, driver, "provider") @@ -599,7 +599,7 @@ async def test_the_first_row_leaves_the_agent_running_as_this_machine( await keeps(app, driver) await keeps(app, driver) - assert app._models == [Runs("claude/claude-opus-5:high")] + assert app._models == {"assistant": Runs("claude/claude-opus-5:high")} @pytest.mark.timeout(60) @@ -922,26 +922,6 @@ async def test_the_key_that_used_to_take_an_account_away_takes_nothing_away() -> assert providers.find("claude", "deepseek") is not None -def test_an_agent_is_made_as_the_account_it_was_given() -> None: - """What the sheet answered is a setting of the agent, done to it before the flow starts.""" - from hmz.coganchor.agents import ClaudeCodeAgent, ClaudeCodeAgentConfig - - _account() - app = Humanize() - app._models = [Runs("claude/m:high", "", "", "deepseek")] - made = ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high")) - - (agent,) = app._as_they_were_set_up([made]) - - assert agent.config.provider == "deepseek" - assert agent.config.machine is None # it works here, as it did before - - # And one nobody named an account for is the agent that was made, untouched. - app._models = [Runs("claude/m:high")] - again = ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high")) - assert app._as_they_were_set_up([again]) == [again] - - @pytest.mark.timeout(60) @unittest.mock.patch("hmz.tui.app.installed", return_value=CLAUDE) async def test_an_agent_told_to_run_as_nobody_is_a_line_to_correct( @@ -950,38 +930,35 @@ async def test_an_agent_told_to_run_as_nobody_is_a_line_to_correct( """An agent that cannot find its account must not quietly run as whoever started it.""" app = Humanize() async with app.run_test() as driver: - app._models = [Runs("claude/claude-opus-5:max", "", "", "nonesuch")] + set_up(app, "chat", {"assistant": Runs("claude/claude-opus-5:max", "nonesuch")}) await driver.press(*"go") await driver.press("enter") await until(lambda: "nonesuch" in transcript(app), driver) + await until(lambda: app._run is None, driver) said = transcript(app) - assert "hmz: no claude provider called 'nonesuch'" in said - assert ( - "Traceback" not in said - ) # said at the prompt, rather than raised out of a thread - assert not app._agents # and nothing started + assert "hmz:" in said + # Said at the prompt, rather than raised out of a thread. + assert "Traceback" not in said def test_what_an_agent_runs_as_is_kept_and_read_back(tmp_path: Path) -> None: - """As the anchor is: written only where there is an account to write.""" + """As `-a` writes it: the account after an `@`, and nothing where there is none.""" kept = Settings(tmp_path) kept.remember( "rlar", - ("actor", "reviewer"), - [Runs("claude/m:high", "", "", "deepseek"), Runs("codex/n:low")], + {"actor": Runs("claude/m:high", "deepseek"), "reviewer": Runs("codex/n:low")}, ) - assert Settings(tmp_path).agents("rlar") == [ - Runs("claude/m:high", "", "", "deepseek"), - Runs("codex/n:low"), - ] + assert Settings(tmp_path).agents("rlar") == { + "actor": Runs("claude/m:high", "deepseek"), + "reviewer": Runs("codex/n:low"), + } held = Settings(tmp_path)._read() agents = held["workspaces"][str(tmp_path.resolve())]["flows"]["rlar"]["agents"] - assert agents["actor"]["provider"] == "deepseek" - # An agent nobody named one for says nothing, which is what a file written before there - # were any says too -- and reads back as this machine's own account. - assert "provider" not in agents["reviewer"] + assert agents["actor"] == "claude@deepseek/m:high" + # An agent nobody named one for says nothing -- and reads back as this machine's own. + assert agents["reviewer"] == "codex/n:low" @pytest.mark.timeout(60) diff --git a/tests/integration/tui/test_queued.py b/tests/integration/tui/test_queued.py index 5e51d8a4..9bd34a7c 100644 --- a/tests/integration/tui/test_queued.py +++ b/tests/integration/tui/test_queued.py @@ -23,7 +23,7 @@ from hmz.tui.app import _PINNED from hmz.tui.monitor import short from tests.stubs import ShellAgent, ShellSession, written -from tests.tui.fixtures import transcript +from tests.tui.fixtures import holding, set_up, transcript if TYPE_CHECKING: from collections.abc import Callable @@ -59,17 +59,21 @@ def new(self, cwd: str | os.PathLike[str] | None = None) -> Steerable: #: A flow that runs until a file appears, so that a line can be typed while it is up and the #: flow can then be let finish of its own accord. FLOW = """ -import time +import asyncio from pathlib import Path -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow + +class Agents(AgentCollection): + coder: Agent -@flow -def run(agents: tuple[AgentBase], task: str) -> None: + +@flow(agents=Agents, envs=EnvCollection, params=FlowParams, name="flow") +async def run(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: while not Path("go.txt").exists(): - time.sleep(0.02) + await asyncio.sleep(0.02) """ @@ -94,8 +98,9 @@ async def _running(app: Humanize, driver: Pilot[None]) -> None: Which is a flow between two turns, or one inside a sleep of its own -- the moment a line has nowhere to go but the queue. """ - app._flow_named, app._models = "flow", [Runs("claude/m:high")] - app._agents = [ShellAgent(CONFIG)] + app._flow_named, app._models = "flow", {"coder": Runs("claude/m:high")} + app._declared = None # a flow nothing here loads, whose one role is `coder` + holding(app, ShellAgent(CONFIG)) app._queued = [] await driver.pause() @@ -410,17 +415,17 @@ async def test_what_a_flow_that_ended_never_took_is_said_to_have_been_dropped( """ app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await driver.press(*"the task") await driver.press("enter") - await until(lambda: bool(app._agents), driver) + await until(lambda: app._run is not None, driver) await driver.press(*"and this too") await driver.press("enter") await until(lambda: bool(_pinned(app)), driver) (waiting / "go.txt").write_text("") # and the flow runs out of things to do - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None, driver) await until(lambda: "never sent" in transcript(app), driver) assert _pinned(app) == "" @@ -644,7 +649,7 @@ async def test_three_lines_typed_in_a_row_are_three_turns_of_a_chat() -> None: app = Humanize() async with app.run_test() as driver: await _running(app, driver) - agent = app._agents[0] + generation = app._generation prompts: list[str] = [] def chatting() -> None: @@ -652,7 +657,7 @@ def chatting() -> None: while said: # What a session does on the way in, then what the flow does after. prompts.append("\n\n".join([said, *app._at_turn_start()])) - said = app._listen(agent) + said = app._listen(generation) talking = threading.Thread(target=chatting) talking.start() @@ -662,7 +667,7 @@ def chatting() -> None: await driver.press("enter") await until(lambda: len(prompts) == 4, driver) finally: - app._agents = [] # which is what stopping the flow leaves behind + app._run = None # which is what stopping the flow leaves behind app._spoke.set() talking.join(5) @@ -708,6 +713,4 @@ async def test_a_pinned_line_is_cut_to_what_is_left_beside_it() -> None: beside = app.query_one("#above", Static).region assert _pinned(app).endswith("…") assert pin.right <= beside.x # cut short of it rather than over it - assert beside.width >= len( - "assistant · claude/m:high" - ) # which still fits whole + assert beside.width >= len("coder · claude/m:high") # which still fits whole diff --git a/tests/integration/tui/test_quick_flow.py b/tests/integration/tui/test_quick_flow.py index 4870347e..094c82c9 100644 --- a/tests/integration/tui/test_quick_flow.py +++ b/tests/integration/tui/test_quick_flow.py @@ -21,10 +21,10 @@ from hmz.tui import Humanize from hmz.tui.app import _COMMANDS, Editor from hmz.tui.complete import offered -from hmz.tui.pick import _SAVE, Flows, Unbounded +from hmz.tui.pick import _BUDGET, _SAVE, Configures, Flows from tests.integration.tui.test_app import opens from tests.stubs import ShellAgent, written -from tests.tui.fixtures import transcript, until +from tests.tui.fixtures import holding, transcript, until if TYPE_CHECKING: from pathlib import Path @@ -37,49 +37,87 @@ #: A flow of one agent, for the workspace that has set none of them up yet. _ONE = """ -from hmz._legacy_flows import Agent, flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0].new()(task) +class Agents(AgentCollection): + worker: Agent + + +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def loop(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: + pass """ #: A flow of two agents, for what happens to a workspace set up for one of them and then #: handed a flow that wants both. _PAIR = """ -from typing import NamedTuple - -from hmz._legacy_flows import Agent, flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow -class Pair(NamedTuple): +class Pair(AgentCollection): builder: Agent reviewer: Agent -@flow -def run(agents: Pair, task: str) -> None: - agents.builder.new()(task) +@flow(agents=Pair, envs=EnvCollection, params=FlowParams) +async def pair(task: str, *, agents: Pair, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: + pass """ -#: A file holding a flow of its own name and another beside it whose name has a dash in it -- -#: which `humanize1:gen-idea` is, and which a sigil that stopped reading at the dash would put -#: to the conversation whole instead of running. +#: A flow that takes params, for a workspace whose kept ones it no longer accepts. +_SETTABLE = """ +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow + + +class Agents(AgentCollection): + worker: Agent + + +class Params(FlowParams): + rounds: int = 3 + + +@flow(agents=Agents, envs=EnvCollection, params=Params) +async def settable(task: str, *, agents: Agents, envs: EnvCollection, params: Params, + ctx: FlowContext) -> None: + pass +""" + +#: A module holding a flow of its own name and another beside it whose name has a dash in it +#: -- which `humanize1:gen-idea` is, and which a sigil that stopped reading at the dash would +#: put to the conversation whole instead of running. _PHASES = """ -from hmz._legacy_flows import Agent, flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow + + +class Agents(AgentCollection): + worker: Agent -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0].new()(task) +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def phases(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: + pass -@flow(name="gen-idea") -def idea(agents: tuple[Agent], task: str) -> None: - agents[0].new()(task) +@flow(agents=Agents, envs=EnvCollection, params=FlowParams, name="gen-idea") +async def idea(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: + pass """ +#: What a run of a flow may spend, which every flow but `chat` is given before it is run. +_SPENDS = {"cost": 1.0} + +#: What `chat` is set up with here, which is its one agent role and no budget at all. +_CHAT = {"assistant": Runs("claude/m:high")} + +#: What each flow of one agent role is set up with here. +_WORKER = {"worker": Runs("claude/m:high")} + @pytest.fixture def backend(monkeypatch: pytest.MonkeyPatch) -> None: @@ -91,21 +129,26 @@ def backend(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.setattr(hmz.tui.pick, "installed", lambda: dict(_INSTALLED)) +#: One run that would have been started: the flow, what its roles run, and the task. +type Started = tuple[str, dict[str, Runs], str] + + @pytest.fixture -def started(monkeypatch: pytest.MonkeyPatch) -> list[list[str]]: - """Catches the command line each run would have been started on, and starts none. +def started(monkeypatch: pytest.MonkeyPatch) -> list[Started]: + """Catches each run that would have been started, and starts none. A run is a backend, a thread and a process; what these are about is which flow got - started and on what, which is the line `hmz exec` would have been handed. + started and on what, which is what the runtime would have been handed. """ - lines: list[list[str]] = [] + runs: list[Started] = [] - def caught(_self: Humanize, argv: list[str], resume: object = None) -> None: - """What starting a flow comes to here, which is writing down the line.""" - lines.append(argv) + def caught(self: Humanize, task: str, resume: object = None) -> None: + """What starting a flow comes to here, which is writing down what it was.""" + del resume + runs.append((self._flow_named, dict(self._models), task)) monkeypatch.setattr(Humanize, "_flow", caught) - return lines + return runs async def sends(app: Humanize, driver: Pilot[None], line: str) -> None: @@ -139,33 +182,36 @@ async def saves(app: Humanize, driver: Pilot[None]) -> None: """ sheet = cast("Flows", app.screen) await until(lambda: sheet._inside, driver) + # A run of it is given a budget, which none of these flows has been yet: set on its row, + # as a duration, before the menu is saved. + await opens(app, driver, _BUDGET) + await until(lambda: isinstance(app.screen, Configures), driver) + await driver.press(*"1h") + await driver.press("enter") + await until(lambda: app.screen is sheet, driver) await opens(app, driver, _SAVE) - # None of these flows declares a budget and none of these tests sets one, so saving asks - # whether a run with nothing at all to stop it is what was meant. It is, here. - if isinstance(app.screen, Unbounded): - await driver.press("enter") await until(lambda: not isinstance(app.screen, Flows), driver) @pytest.mark.timeout(60) async def test_a_flow_this_workspace_has_set_up_runs_on_the_line_that_named_it( - tmp_path: Path, started: list[list[str]] + tmp_path: Path, started: list[Started] ) -> None: """The whole point: two answers already given are not two answers to give again.""" - Settings(tmp_path).remember("chat", ("assistant",), [Runs("claude/m:high")]) + Settings(tmp_path).remember("chat", _CHAT) app = Humanize() async with app.run_test() as driver: await sends(app, driver, "$chat fix the build") await until(lambda: bool(started), driver) - assert started == [["-f", "chat", "-a", "claude/m:high", "fix the build"]] + assert started == [("chat", _CHAT, "fix the build")] assert not isinstance(app.screen, Flows) # no menu at all assert "$chat fix the build" in transcript(app) @pytest.mark.timeout(90) async def test_a_flow_never_set_up_here_opens_the_menu_and_runs_once_it_is_saved( - tmp_path: Path, backend: None, started: list[list[str]] + tmp_path: Path, backend: None, started: list[Started] ) -> None: """A flow with no agents chosen for it is a flow that stops on its first turn.""" written(tmp_path / ".humanize" / "flows", "loop", _ONE) @@ -181,7 +227,7 @@ async def test_a_flow_never_set_up_here_opens_the_menu_and_runs_once_it_is_saved await saves(app, driver) await until(lambda: bool(started), driver) - assert started == [["-f", "local/loop", "-a", "claude/m:high", "fix the build"]] + assert started == [("local/loop", _WORKER, "fix the build")] assert Settings(tmp_path).flow == "local/loop" # and it is set up now # And so the same line a second time is the run, with no menu in the way. @@ -192,12 +238,12 @@ async def test_a_flow_never_set_up_here_opens_the_menu_and_runs_once_it_is_saved await until(lambda: bool(started), driver) assert not isinstance(again.screen, Flows) - assert started == [["-f", "local/loop", "-a", "claude/m:high", "fix the build"]] + assert started == [("local/loop", _WORKER, "fix the build")] @pytest.mark.timeout(60) async def test_a_flow_that_grew_an_agent_is_asked_about_rather_than_run_short_of_one( - tmp_path: Path, backend: None, started: list[list[str]] + tmp_path: Path, backend: None, started: list[Started] ) -> None: """What was remembered is one agent, and the flow drives two: a place with nobody in it. @@ -205,7 +251,9 @@ async def test_a_flow_that_grew_an_agent_is_asked_about_rather_than_run_short_of nothing against it rather than the builder's model quietly moved along one. """ written(tmp_path / ".humanize" / "flows", "pair", _PAIR) - Settings(tmp_path).remember("local/pair", ("builder",), [Runs("claude/m:high")]) + Settings(tmp_path).remember( + "local/pair", {"builder": Runs("claude/m:high")}, budget=_SPENDS + ) app = Humanize() async with app.run_test() as driver: await sends(app, driver, "$local/pair fix the build") @@ -216,15 +264,19 @@ async def test_a_flow_that_grew_an_agent_is_asked_about_rather_than_run_short_of @pytest.mark.timeout(60) async def test_settings_the_flow_no_longer_accepts_are_asked_again_rather_than_dropped( - tmp_path: Path, backend: None, started: list[list[str]] + tmp_path: Path, backend: None, started: list[Started] ) -> None: - """A flow that renamed a setting under what was written down for it is one to answer.""" + """A flow that renamed a param under what was written down for it is one to answer.""" + written(tmp_path / ".humanize" / "flows", "settable", _SETTABLE) Settings(tmp_path).remember( - "ralph_loop", ("",), [Runs("claude/m:high")], {"nothing-of-the-sort": 1} + "local/settable", + _WORKER, + params={"nothing-of-the-sort": 1}, + budget=_SPENDS, ) app = Humanize() async with app.run_test() as driver: - await sends(app, driver, "$ralph_loop fix the build") + await sends(app, driver, "$local/settable fix the build") await until(lambda: isinstance(app.screen, Flows), driver) assert not started @@ -232,7 +284,7 @@ async def test_settings_the_flow_no_longer_accepts_are_asked_again_rather_than_d @pytest.mark.timeout(60) async def test_walking_out_of_the_menu_starts_nothing_and_says_so( - tmp_path: Path, backend: None, started: list[list[str]] + tmp_path: Path, backend: None, started: list[Started] ) -> None: """A line typed to start something must not vanish without a word about it.""" written(tmp_path / ".humanize" / "flows", "loop", _ONE) @@ -249,7 +301,7 @@ async def test_walking_out_of_the_menu_starts_nothing_and_says_so( @pytest.mark.timeout(60) async def test_a_flow_that_is_not_there_is_a_line_to_correct_and_not_the_end( - started: list[list[str]], + started: list[Started], ) -> None: """Said the way `/nosuchcommand` is: the sigil was meant, the name after it is the typo.""" app = Humanize() @@ -272,7 +324,7 @@ async def test_a_flow_that_is_not_there_is_a_line_to_correct_and_not_the_end( ], ) async def test_a_line_that_merely_begins_with_a_dollar_is_still_a_line( - line: str, started: list[list[str]] + line: str, started: list[Started] ) -> None: """`$` is a sigil on a name, so a `$` with no name after it is eaten by nothing.""" app = Humanize() @@ -289,10 +341,10 @@ async def test_a_line_that_merely_begins_with_a_dollar_is_still_a_line( @pytest.mark.timeout(60) async def test_a_prompt_written_under_the_name_is_the_prompt( - tmp_path: Path, started: list[list[str]] + tmp_path: Path, started: list[Started] ) -> None: """A long prompt is broken over several lines, and the `$` is still on the first of them.""" - Settings(tmp_path).remember("chat", ("assistant",), [Runs("claude/m:high")]) + Settings(tmp_path).remember("chat", _CHAT) app = Humanize() async with app.run_test() as driver: await driver.press(*"$chat") @@ -301,24 +353,22 @@ async def test_a_prompt_written_under_the_name_is_the_prompt( await driver.press("enter") await until(lambda: bool(started), driver) - assert started == [["-f", "chat", "-a", "claude/m:high", "fix the build"]] + assert started == [("chat", _CHAT, "fix the build")] @pytest.mark.timeout(60) async def test_one_of_the_several_flows_a_file_holds_is_named_dash_and_all( - tmp_path: Path, started: list[list[str]] + tmp_path: Path, started: list[Started] ) -> None: """`:` is a name like any other, and what is inside may be called anything.""" written(tmp_path / ".humanize" / "flows", "phases", _PHASES) - Settings(tmp_path).remember("local/phases:gen-idea", ("",), [Runs("claude/m:high")]) + Settings(tmp_path).remember("local/phases:gen-idea", _WORKER, budget=_SPENDS) app = Humanize() async with app.run_test() as driver: await sends(app, driver, "$local/phases:gen-idea fix the build") await until(lambda: bool(started), driver) - assert started == [ - ["-f", "local/phases:gen-idea", "-a", "claude/m:high", "fix the build"] - ] + assert started == [("local/phases:gen-idea", _WORKER, "fix the build")] @pytest.mark.timeout(60) @@ -326,7 +376,7 @@ async def test_nothing_is_offered_against_a_dollar_while_an_agent_waits_to_be_an tmp_path: Path, ) -> None: """The next line typed is the answer, whatever it begins with, so enter must send it.""" - Settings(tmp_path).remember("chat", ("assistant",), [Runs("claude/m:high")]) + Settings(tmp_path).remember("chat", _CHAT) app = Humanize() async with app.run_test() as driver: app._asking = Question("which way?") @@ -341,28 +391,28 @@ async def test_nothing_is_offered_against_a_dollar_while_an_agent_waits_to_be_an @pytest.mark.timeout(60) async def test_a_dollar_while_a_flow_runs_is_refused_the_way_choosing_one_is( - tmp_path: Path, started: list[list[str]] + tmp_path: Path, started: list[Started] ) -> None: """Choosing a flow is shut while one runs, and this is choosing a flow.""" - Settings(tmp_path).remember("chat", ("assistant",), [Runs("claude/m:high")]) + Settings(tmp_path).remember("chat", _CHAT) app = Humanize() async with app.run_test() as driver: - app._agents = [ShellAgent(AgentConfig(model="m", effort="high"))] + run = holding(app, ShellAgent(AgentConfig(model="m", effort="high"))) await sends(app, driver, "$chat fix the build") await until(lambda: "a flow is running" in transcript(app), driver) - assert app._agents # left exactly as it was + assert app._run is run # left exactly as it was assert not started @pytest.mark.timeout(60) async def test_a_dollar_naming_a_flow_and_nothing_else_chooses_it_and_waits( - tmp_path: Path, started: list[list[str]] + tmp_path: Path, started: list[Started] ) -> None: """With nothing said after the name there is nothing to start on, so nothing starts.""" kept = Settings(tmp_path) - kept.remember("chat", ("assistant",), [Runs("claude/m:high")]) - kept.remember("ralph_loop", ("",), [Runs("claude/m:high")]) + kept.remember("chat", _CHAT) + kept.remember("ralph_loop", _WORKER, budget=_SPENDS) app = Humanize() assert app._flow_named == "ralph_loop" # the one this workspace was last run with async with app.run_test() as driver: @@ -403,3 +453,34 @@ async def test_tab_takes_the_flow_that_is_offered_under_the_sigil() -> None: await driver.pause() assert app.query_one(Editor).text == "$chat " + + +@pytest.mark.timeout(60) +async def test_an_environment_role_the_flow_no_longer_declares_is_not_handed_to_it( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """A role renamed since is one no row can clear, and must not refuse every later run.""" + from typing import Any + + from hmz.runtime.doing.core import Hmz + + handed: list[dict[str, Any]] = [] + + def caught(self: Hmz, flow: str, task: str, **said: Any) -> None: + del self, flow, task + handed.append(said) + raise RuntimeError("caught") + + monkeypatch.setattr(Hmz, "run", caught) + written(tmp_path / ".humanize" / "flows", "loop", _ONE) + Settings(tmp_path).remember( + "local/loop", _WORKER, envs={"scratch": "local@/tmp"}, budget=_SPENDS + ) + app = Humanize() + async with app.run_test() as driver: + await sends(app, driver, "$local/loop fix the build") + await until(lambda: bool(handed), driver) + + (said,) = handed + assert said["envs"] == {} + assert said["agents"] == {"worker": "claude/m:high"} diff --git a/tests/integration/tui/test_resume.py b/tests/integration/tui/test_resume.py index ecffa91e..df4346a7 100644 --- a/tests/integration/tui/test_resume.py +++ b/tests/integration/tui/test_resume.py @@ -22,26 +22,31 @@ from hmz.runtime.epic import epics, state from hmz.tui import Humanize from tests.stubs import written -from tests.tui.fixtures import transcript, until +from tests.tui.fixtures import Holding, holding, transcript, until if TYPE_CHECKING: from pathlib import Path from textual.pilot import Pilot -#: A flow that says it can be picked up, and counts the runs of itself in what it is handed. +#: A flow that says it can be picked up, and counts the runs of itself in what it keeps. COUNTS = '''"""Counts the runs of itself.""" from pathlib import Path -from typing import Any -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow -@flow(resumable=True) -def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None: - state["rounds"] = state.get("rounds", 0) + 1 +class Agents(AgentCollection): + worker: Agent + + +@flow(agents=Agents, envs=EnvCollection, params=FlowParams, resumable=True) +async def counts(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: + state = ctx.state + assert state is not None + state["rounds"] = (state["rounds"] if "rounds" in state else 0) + 1 Path("rounds.txt").write_text(str(state["rounds"])) ''' @@ -50,45 +55,17 @@ def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None: from pathlib import Path -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - Path("plain.txt").write_text(task) -''' - -#: One that says it can be picked up and writes nothing down, which is what a run stopped -#: before it got anywhere leaves behind: the mark, and nothing under it. -BLANK = '''"""Says it can be picked up, and never writes down where it got to.""" - -from pathlib import Path -from typing import Any - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow(resumable=True) -def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None: - Path("blank.txt").write_text(task) -''' - -#: One that says it can be picked up, writes something down and then empties it, which is a -#: flow saying the next run here starts clean rather than carrying this one on. -EMPTIES = '''"""Writes down where it got to, and then says the next run starts clean.""" +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams, flow -from typing import Any -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +class Agents(AgentCollection): + worker: Agent -@flow(resumable=True) -def run(agents: tuple[AgentBase], task: str, state: dict[str, Any]) -> None: - state["rounds"] = state.get("rounds", 0) + 1 - state.clear() +@flow(agents=Agents, envs=EnvCollection, params=FlowParams) +async def plain(task: str, *, agents: Agents, envs: EnvCollection, params: FlowParams, + ctx: FlowContext) -> None: + Path("plain.txt").write_text(task) ''' #: A `claude` that answers whatever it is told, since what is being tested is the run rather @@ -117,8 +94,6 @@ def workspace(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: where.mkdir(parents=True) written(where, "counts", COUNTS) written(where, "plain", PLAIN) - written(where, "blank", BLANK) - written(where, "empties", EMPTIES) monkeypatch.chdir(tmp_path) return tmp_path @@ -126,15 +101,12 @@ def workspace(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: def _ran(flow: str, task: str) -> None: """Runs one flow here, the way a command line would, so there is a run to carry on. - On the fake `claude` this suite puts on PATH rather than on a stand-in of our own: a run - is carried on as a command line naming what each of its agents runs, so the agents of a - run have to be agents something can name. + On the fake `claude` this suite puts on PATH: a run is carried on as the specs each of its + roles was given, so the agents of a run have to be agents something can name. """ - from hmz.coganchor.agents import driver - from hmz.runtime.runner import Runner + from hmz.runtime import Hmz - agent, config = driver("claude") - Runner(flow, [agent(config(model="m", effort="high"))]).run(task) + Hmz().run(flow, task, agents={"worker": "claude/m:high"}, budget={"cost": 1}).run() async def _resumes(app: Humanize, driver: Pilot[None]) -> None: @@ -184,17 +156,17 @@ async def test_a_directory_nothing_has_been_run_in_says_so(workspace: Path) -> N @pytest.mark.timeout(60) -async def test_a_flow_that_no_longer_says_it_can_be_picked_up_says_why( +async def test_a_run_of_a_flow_that_cannot_be_picked_up_is_not_the_one_carried_on( workspace: Path, ) -> None: - """Asked of the flow, as it is wherever it is asked -- and named, so it can be fixed.""" + """A conversation had since is not a run to carry on, and says why there is none.""" _ran("plain", "go") app = Humanize() async with app.run_test() as driver: await _resumes(app, driver) - assert "plain does not say it can be picked up" in transcript(app) + assert "no run here was of a flow that can be picked up" in transcript(app) assert len(epics(workspace)) == 1 # and nothing was started @@ -208,8 +180,12 @@ async def test_a_run_that_left_nothing_behind_is_not_carried_on( on from the day before yesterday, because yesterday's died before it wrote anything, is a day's work thrown away without anybody being told. """ + from hmz.runtime.epic import RESUME + _ran("counts", "keep going") # which left something - _ran("blank", "go") # and which is not what the last run here was + _ran("counts", "again") # a run from the top, killed before it wrote anything down + (last,) = [one for one in epics(workspace) if (one / RESUME).is_file()][-1:] + (last / RESUME).unlink() app = Humanize() async with app.run_test() as driver: @@ -217,7 +193,7 @@ async def test_a_run_that_left_nothing_behind_is_not_carried_on( assert "left nothing behind" in transcript(app) assert len(epics(workspace)) == 2 - assert (workspace / "rounds.txt").read_text() == "1" + assert (workspace / "rounds.txt").read_text() == "1" # and nothing ran again @pytest.mark.timeout(60) @@ -240,27 +216,6 @@ async def test_a_record_that_cannot_be_read_back_says_so(workspace: Path) -> Non assert "cannot be read back" in transcript(app) -@pytest.mark.timeout(60) -async def test_a_run_that_emptied_what_it_wrote_is_not_carried_on( - workspace: Path, -) -> None: - """A flow that cleared its state said the next run here starts clean. - - Which is the opposite of what handing it that state back would say. A run stopped for - having spent its allowance is not that: it was stopped rather than finished, so it leaves - what it kept and picking it up is the point of having kept it. - """ - _ran("empties", "go") - - app = Humanize() - async with app.run_test() as driver: - await _resumes(app, driver) - - assert "left nothing behind" in transcript(app) - assert "starts from the top" in transcript(app) - assert len(epics(workspace)) == 1 - - @pytest.mark.timeout(60) async def test_a_flow_marked_since_the_run_is_asked_of_the_flow( workspace: Path, @@ -290,13 +245,13 @@ async def test_carrying_on_is_refused_while_a_flow_is_running(workspace: Path) - app = Humanize() async with app.run_test() as driver: - app._agents = [ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high"))] + holding(app, ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high"))) await _resumes(app, driver) assert "no picking a run up while a flow is running" in transcript(app) assert "ctrl+c twice stops it first" in transcript(app) assert len(epics(workspace)) == 1 - app._agents = [] + app._run = None @pytest.mark.timeout(60) @@ -309,22 +264,18 @@ async def test_carrying_on_is_refused_while_the_flow_is_still_unwinding( a round the stopped run had already recorded. And `ctrl+c twice` is not the answer here, since that is what was just pressed. """ - from hmz.coganchor.agents.claude import ClaudeCodeAgent, ClaudeCodeAgentConfig - _ran("counts", "keep going") app = Humanize() async with app.run_test() as driver: - # What `ctrl+c` twice leaves behind: let go of as the running agents, kept as the - # ones on their way out. - app._stopping = [ - ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high")) - ] + # What `ctrl+c` twice leaves behind: let go of as the run going, kept as the one on + # its way out. + app._stopping = Holding() await _resumes(app, driver) assert "while the flow is still stopping" in transcript(app) assert len(epics(workspace)) == 1 - app._stopping = [] + app._stopping = None @pytest.mark.timeout(60) diff --git a/tests/integration/tui/test_settings.py b/tests/integration/tui/test_settings.py index e1891f70..3c02a90f 100644 --- a/tests/integration/tui/test_settings.py +++ b/tests/integration/tui/test_settings.py @@ -21,77 +21,81 @@ def test_a_workspace_that_has_run_nothing_remembers_nothing(tmp_path: Path) -> N kept = Settings(tmp_path) assert kept.flow == "" - assert kept.agents("chat") == [] + assert kept.agents("chat") == {} def test_what_was_set_up_is_what_is_read_back(tmp_path: Path) -> None: """The whole point: a project driven by two agents is driven by them again tomorrow.""" Settings(tmp_path).remember( "rlar", - ("actor", "reviewer"), - [Runs("claude/claude-opus-5:high"), Runs("codex/gpt-5.6-sol:xhigh")], + { + "actor": Runs("claude/claude-opus-5:high"), + "reviewer": Runs("codex/gpt-5.6-sol:xhigh"), + }, ) # A second one, as opening the interface again is. again = Settings(tmp_path) assert again.flow == "rlar" - assert again.agents("rlar") == [ - Runs("claude/claude-opus-5:high"), - Runs("codex/gpt-5.6-sol:xhigh"), - ] + assert again.agents("rlar") == { + "actor": Runs("claude/claude-opus-5:high"), + "reviewer": Runs("codex/gpt-5.6-sol:xhigh"), + } + +def test_an_agent_is_kept_under_its_role_as_the_word_a_command_line_takes( + tmp_path: Path, +) -> None: + """So that a flow which grows a role does not hand the reviewer's model to the builder. -def test_an_agent_is_kept_under_what_its_flow_calls_it(tmp_path: Path) -> None: - """So that a flow which grows an agent does not hand the reviewer's model to the builder.""" + And written as `-a` writes it after the role, so what the menu keeps and what a command + line says are one word: the account after an `@`, and a model holding slashes of its own + surviving the round trip. + """ Settings(tmp_path).remember( - "rlar", ("actor", "reviewer"), [Runs("claude/m:high"), Runs("codex/n:low")] + "rlar", + {"actor": Runs("claude/m:high", "work"), "reviewer": Runs("codex/n:low")}, ) - Settings(tmp_path).remember("chat", ("",), [Runs("kimi/kimi-code/k3:max")]) + Settings(tmp_path).remember("chat", {"assistant": Runs("kimi/kimi-code/k3:max")}) held = yaml.safe_load((home() / "settings.yaml").read_text()) flows = held["workspaces"][str(tmp_path.resolve())]["flows"] - assert list(flows["rlar"]["agents"]) == ["actor", "reviewer"] - assert flows["rlar"]["agents"]["reviewer"] == { - "cli": "codex", - "model": "n", - "effort": "low", - "goals": True, - # Written down as the nothing it is: nobody was asked whether this one may search, - # and a `True` here would be an answer the file made up on their behalf. - "web_search": None, + assert flows["rlar"]["agents"] == { + "actor": "claude@work/m:high", + "reviewer": "codex/n:low", + } + assert flows["chat"]["agents"] == {"assistant": "kimi/kimi-code/k3:max"} + assert Settings(tmp_path).agents("chat") == { + "assistant": Runs("kimi/kimi-code/k3:max") } - # A flow that says only how many it drives has nothing to call them, so they are - # numbered -- and a model holding slashes of its own survives the round trip. - assert list(flows["chat"]["agents"]) == ["1"] - assert flows["chat"]["agents"]["1"]["model"] == "kimi-code/k3" + assert Settings(tmp_path).agents("rlar")["actor"] == Runs("claude/m:high", "work") def test_each_flow_of_a_workspace_is_kept_beside_the_others(tmp_path: Path) -> None: """What an agent runs only means anything against the flow that drives it.""" kept = Settings(tmp_path) - kept.remember("chat", ("",), [Runs("claude/m:high")]) - kept.remember("ralph_loop", ("",), [Runs("codex/n:low")]) + kept.remember("chat", {"assistant": Runs("claude/m:high")}) + kept.remember("ralph_loop", {"coder": Runs("codex/n:low")}) again = Settings(tmp_path) assert again.flow == "ralph_loop" # the one it was last run with - assert again.agents("chat") == [ - Runs("claude/m:high") - ] # and the other is still there - assert again.agents("ralph_loop") == [Runs("codex/n:low")] + # And the other is still there. + assert again.agents("chat") == {"assistant": Runs("claude/m:high")} + assert again.agents("ralph_loop") == {"coder": Runs("codex/n:low")} def test_one_workspace_does_not_take_anothers(tmp_path: Path) -> None: (mine := tmp_path / "mine").mkdir() (theirs := tmp_path / "theirs").mkdir() - Settings(mine).remember("chat", ("",), [Runs("claude/m:high")]) + Settings(mine).remember("chat", {"assistant": Runs("claude/m:high")}) Settings(theirs).remember( - "rlar", ("a", "b"), [Runs("codex/n:low"), Runs("codex/n:low")] + "rlar", {"a": Runs("codex/n:low"), "b": Runs("codex/n:low")} ) assert Settings(mine).flow == "chat" - assert Settings(mine).agents("chat") == [Runs("claude/m:high")] + assert Settings(mine).agents("chat") == {"assistant": Runs("claude/m:high")} @pytest.mark.parametrize( @@ -108,11 +112,28 @@ def test_a_file_that_is_not_one_is_a_workspace_with_nothing_remembered( kept = Settings(tmp_path) assert kept.flow == "" - assert kept.agents("chat") == [] - kept.remember( - "chat", ("",), [Runs("claude/m:high")] - ) # and it is written over rather than kept - assert Settings(tmp_path).agents("chat") == [Runs("claude/m:high")] + assert kept.agents("chat") == {} + # And it is written over rather than kept. + kept.remember("chat", {"assistant": Runs("claude/m:high")}) + assert Settings(tmp_path).agents("chat") == {"assistant": Runs("claude/m:high")} + + +def test_an_agent_an_older_humanize_wrote_down_reads_as_nothing_remembered( + tmp_path: Path, +) -> None: + """One written as the fields it had then is a flow to be asked about again. + + Rather than half an answer -- which role it was for is the half this cannot guess. + """ + Settings(tmp_path).remember("rlar", {"actor": Runs("claude/m:high")}) + where = home() / "settings.yaml" + held = yaml.safe_load(where.read_text()) + held["workspaces"][str(tmp_path.resolve())]["flows"]["rlar"]["agents"] = { + "actor": {"cli": "claude", "model": "m", "effort": "high", "goals": True} + } + where.write_text(yaml.safe_dump(held, sort_keys=False)) + + assert Settings(tmp_path).agents("rlar") == {} def test_a_home_that_cannot_be_written_is_not_a_reason_to_stop(tmp_path: Path) -> None: @@ -120,90 +141,97 @@ def test_a_home_that_cannot_be_written_is_not_a_reason_to_stop(tmp_path: Path) - home().mkdir(parents=True, exist_ok=True) home().chmod(0o500) try: - Settings(tmp_path).remember("chat", ("",), [Runs("claude/m:high")]) + Settings(tmp_path).remember("chat", {"assistant": Runs("claude/m:high")}) finally: home().chmod(0o700) -def test_where_an_agent_works_is_kept_with_what_it_runs(tmp_path: Path) -> None: - """So that a project driven against a container is driven against it again tomorrow.""" +def test_where_an_environment_is_kept_beside_what_the_agents_run( + tmp_path: Path, +) -> None: + """So that a project driven on a machine of its own is driven on it again tomorrow.""" Settings(tmp_path).remember( "rlar", - ("actor", "reviewer"), - [Runs("claude/m:high", "ssh://box"), Runs("codex/n:low")], + {"actor": Runs("claude/m:high")}, + {"repo": "ssh@box/home/me/repo"}, ) held = yaml.safe_load((home() / "settings.yaml").read_text()) - agents = held["workspaces"][str(tmp_path.resolve())]["flows"]["rlar"]["agents"] - assert agents["actor"]["anchor"] == "ssh://box" - # An agent that works here says nothing about a machine, which is what a file written - # before there were any also says. - assert "anchor" not in agents["reviewer"] + flows = held["workspaces"][str(tmp_path.resolve())]["flows"] + assert flows["rlar"]["envs"] == {"repo": "ssh@box/home/me/repo"} + assert Settings(tmp_path).envs("rlar") == {"repo": "ssh@box/home/me/repo"} + + # Choosing the agents again says nothing about where they work, so it changes nothing. + Settings(tmp_path).remember("rlar", {"actor": Runs("codex/n:low")}) + assert Settings(tmp_path).envs("rlar") == {"repo": "ssh@box/home/me/repo"} - assert Settings(tmp_path).agents("rlar") == [ - Runs("claude/m:high", "ssh://box"), - Runs("codex/n:low"), - ] + # And an empty one is the way to say none, which erases it. + Settings(tmp_path).remember("rlar", {"actor": Runs("codex/n:low")}, {}) + assert Settings(tmp_path).envs("rlar") == {} def test_how_a_flow_was_set_up_is_kept_beside_what_its_agents_run( tmp_path: Path, ) -> None: - """A flow of twenty settings is not one to answer again every morning.""" + """A flow of twenty params is not one to answer again every morning.""" Settings(tmp_path).remember( - "humanize1", ("builder",), [Runs("claude/m:high")], {"max": 12, "rlcr": True} + "humanize1", + {"builder": Runs("claude/m:high")}, + params={"max": 12, "rlcr": True}, ) again = Settings(tmp_path) - assert again.config("humanize1") == {"max": 12, "rlcr": True} + assert again.params("humanize1") == {"max": 12, "rlcr": True} held = yaml.safe_load((home() / "settings.yaml").read_text()) flows = held["workspaces"][str(tmp_path.resolve())]["flows"] - assert flows["humanize1"]["config"] == {"max": 12, "rlcr": True} + assert flows["humanize1"]["params"] == {"max": 12, "rlcr": True} -def test_choosing_the_agents_again_is_not_a_way_of_forgetting_the_settings( +def test_choosing_the_agents_again_is_not_a_way_of_forgetting_the_params( tmp_path: Path, ) -> None: - """`/agents` says nothing about how the flow itself was set up, so it changes nothing.""" + """Choosing the agents says nothing about how the flow itself was set up.""" Settings(tmp_path).remember( - "humanize1", ("builder",), [Runs("claude/m:high")], {"max": 12} + "humanize1", {"builder": Runs("claude/m:high")}, params={"max": 12} ) - Settings(tmp_path).remember("humanize1", ("builder",), [Runs("codex/n:low")]) + Settings(tmp_path).remember("humanize1", {"builder": Runs("codex/n:low")}) - assert Settings(tmp_path).config("humanize1") == {"max": 12} + assert Settings(tmp_path).params("humanize1") == {"max": 12} def test_a_flow_that_takes_no_setting_up_keeps_nothing(tmp_path: Path) -> None: """Which is most of them, and is what a settings file written before this also says.""" - Settings(tmp_path).remember("chat", ("",), [Runs("claude/m:high")]) + Settings(tmp_path).remember("chat", {"assistant": Runs("claude/m:high")}) - assert Settings(tmp_path).config("chat") == {} + assert Settings(tmp_path).params("chat") == {} + assert Settings(tmp_path).envs("chat") == {} assert Settings(tmp_path).budget("chat") == {} def test_what_a_run_of_a_flow_may_spend_is_kept_beside_how_it_was_set_up( tmp_path: Path, ) -> None: - """Beside the config and not inside it: it is a setting of the run, not of the flow.""" + """Beside the params and not inside them: it is a setting of the run, not of the flow.""" + spends = { + "duration": "PT6H", + "cost": None, + "output_tokens": 10_000, + "graceful": True, + } Settings(tmp_path).remember( "ralph_loop", - ("",), - [Runs("claude/m:high")], - {"rounds": 12}, - {"hours": 6.0, "tokens": 10.0, "dollars": 0.0}, + {"coder": Runs("claude/m:high")}, + params={"rounds": 12}, + budget=spends, ) again = Settings(tmp_path) - assert again.budget("ralph_loop") == { - "hours": 6.0, - "tokens": 10.0, - "dollars": 0.0, - } - assert again.config("ralph_loop") == {"rounds": 12} + assert again.budget("ralph_loop") == spends + assert again.params("ralph_loop") == {"rounds": 12} held = yaml.safe_load((home() / "settings.yaml").read_text()) flows = held["workspaces"][str(tmp_path.resolve())]["flows"] - assert flows["ralph_loop"]["budget"]["hours"] == 6.0 + assert flows["ralph_loop"]["budget"]["duration"] == "PT6H" def test_choosing_the_agents_again_is_not_a_way_of_forgetting_the_budget( @@ -211,23 +239,23 @@ def test_choosing_the_agents_again_is_not_a_way_of_forgetting_the_budget( ) -> None: """The flow's whole entry is replaced, so what is not handed in has to be read back.""" Settings(tmp_path).remember( - "ralph_loop", ("",), [Runs("claude/m:high")], budget={"hours": 6.0} + "ralph_loop", {"coder": Runs("claude/m:high")}, budget={"cost": 6.0} ) - Settings(tmp_path).remember("ralph_loop", ("",), [Runs("codex/n:low")]) + Settings(tmp_path).remember("ralph_loop", {"coder": Runs("codex/n:low")}) - assert Settings(tmp_path).budget("ralph_loop") == {"hours": 6.0} + assert Settings(tmp_path).budget("ralph_loop") == {"cost": 6.0} -def test_a_budget_of_nothing_is_a_flow_back_under_what_it_says_for_itself( - tmp_path: Path, -) -> None: +def test_a_budget_of_nothing_is_forgotten(tmp_path: Path) -> None: """Which is how the menu says a run is no longer held to what was set here.""" Settings(tmp_path).remember( - "ralph_loop", ("",), [Runs("claude/m:high")], budget={"hours": 6.0} + "ralph_loop", {"coder": Runs("claude/m:high")}, budget={"cost": 6.0} ) - Settings(tmp_path).remember("ralph_loop", ("",), [Runs("claude/m:high")], budget={}) + Settings(tmp_path).remember( + "ralph_loop", {"coder": Runs("claude/m:high")}, budget={} + ) assert Settings(tmp_path).budget("ralph_loop") == {} @@ -235,74 +263,18 @@ def test_a_budget_of_nothing_is_a_flow_back_under_what_it_says_for_itself( def test_two_flows_of_one_name_are_two_entries(tmp_path: Path) -> None: """A flow of yours is called by its path, so it cannot inherit a built-in's setup.""" Settings(tmp_path).remember( - "rlar", ("actor",), [Runs("claude/m:high")], {"deep": True} + "rlar", {"actor": Runs("claude/m:high")}, params={"deep": True} ) Settings(tmp_path).remember( - ".humanize/flows/rlar.py", ("actor",), [Runs("codex/n:low")], {"deep": False} + ".humanize/flows/rlar.py", + {"actor": Runs("codex/n:low")}, + params={"deep": False}, ) kept = Settings(tmp_path) - assert kept.config("rlar") == {"deep": True} - assert kept.config(".humanize/flows/rlar.py") == {"deep": False} - assert kept.agents("rlar") == [Runs("claude/m:high")] - - -def test_what_an_agent_may_do_is_kept_and_read_back(tmp_path: Path) -> None: - """As the anchor is: written only where it narrows anything. - - A file written before there was such a setting and one for an agent nobody was asked - about read the same way, which is what leaving it out means. - """ - kept = Settings(tmp_path) - kept.remember( - "rlar", - ("actor", "reviewer"), - [Runs("claude/m:high", "", "read-only"), Runs("codex/n:low")], - ) - - assert Settings(tmp_path).agents("rlar") == [ - Runs("claude/m:high", "", "read-only"), - Runs("codex/n:low"), - ] - held = yaml.safe_load((home() / "settings.yaml").read_text()) - written = held["workspaces"][str(tmp_path.resolve())]["flows"]["rlar"]["agents"] - assert written["actor"]["permission"] == "read-only" - assert "permission" not in written["reviewer"] - - -def test_goal_choices_are_kept_as_on_or_off(tmp_path: Path) -> None: - kept = Settings(tmp_path) - kept.remember( - "rlar", - ("actor", "reviewer"), - [Runs("claude/m:high", goals=False), Runs("codex/n:low", goals=True)], - ) - - assert Settings(tmp_path).agents("rlar") == [ - Runs("claude/m:high", goals=False), - Runs("codex/n:low", goals=True), - ] - held = yaml.safe_load((home() / "settings.yaml").read_text()) - agents = held["workspaces"][str(tmp_path.resolve())]["flows"]["rlar"]["agents"] - assert agents["actor"]["goals"] is False - assert agents["reviewer"]["goals"] is True - - -def test_an_old_entry_takes_the_current_agent_place_suggestion(tmp_path: Path) -> None: - kept = Settings(tmp_path) - kept.remember("rlar", ("actor",), [Runs("claude/m:high")]) - where = home() / "settings.yaml" - held = yaml.safe_load(where.read_text()) - agent = held["workspaces"][str(tmp_path.resolve())]["flows"]["rlar"]["agents"][ - "actor" - ] - del agent["goals"] - where.write_text(yaml.safe_dump(held, sort_keys=False)) - - assert Settings(tmp_path).agents("rlar", (False,)) == [ - Runs("claude/m:high", goals=False) - ] - assert Settings(tmp_path).agents("rlar") == [Runs("claude/m:high", goals=True)] + assert kept.params("rlar") == {"deep": True} + assert kept.params(".humanize/flows/rlar.py") == {"deep": False} + assert kept.agents("rlar") == {"actor": Runs("claude/m:high")} @pytest.mark.timeout(60) @@ -382,7 +354,7 @@ async def test_the_settings_menu_is_two_pages_and_turns_the_reporting_off( monkeypatch.chdir(tmp_path) Settings(tmp_path).answers(enable_sentry=True) - Settings(tmp_path).remember("chat", ("a",), [Runs("claude/m:high")]) + Settings(tmp_path).remember("chat", {"assistant": Runs("claude/m:high")}) app = Humanize() async with app.run_test() as driver: await driver.press(*"/settings") diff --git a/tests/integration/tui/test_stopping.py b/tests/integration/tui/test_stopping.py index f501fbd3..d63c89ae 100644 --- a/tests/integration/tui/test_stopping.py +++ b/tests/integration/tui/test_stopping.py @@ -20,11 +20,10 @@ import pytest from hmz.runtime.epic import epics -from hmz.runtime.kept import Runs from hmz.tui import Humanize from tests.stubs import events as recorded from tests.stubs import written -from tests.tui.fixtures import transcript, until +from tests.tui.fixtures import ONE, Holding, set_up, transcript, until if TYPE_CHECKING: from pathlib import Path @@ -32,18 +31,7 @@ from textual.pilot import Pilot #: A flow that drives one agent for one turn, which is enough to have something to stop. -FLOW = """ -from pathlib import Path - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - session = agents[0].new() - Path("said.txt").write_text(session(task) + "\\n") -""" +FLOW = ONE #: A `claude` that never answers, so that the turn is still open when the line is typed. PATIENT = """ @@ -100,15 +88,12 @@ async def test_a_typed_stop_stops_the_flow_as_the_second_press_does( written(workspace, "flow", FLOW) app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await _typed(driver, "start") - await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), - driver, - ) + await until(lambda: any(agent.sessions for agent in app._agents), driver) await _typed(driver, "/stop") - await until(lambda: not app._agents, driver) + await until(lambda: app._run is None and app._stopping is None, driver) assert "stopping the flow" in transcript(app) (epic,) = epics(workspace) @@ -137,19 +122,15 @@ async def test_a_typed_stop_says_so_where_there_is_nothing_to_stop() -> None: @pytest.mark.timeout(60) async def test_a_flow_already_stopping_is_said_to_be_rather_than_told_again() -> None: - """Telling it again would drop the agents the third press has to reach. + """Telling it again would lose the run the third press has to reach. - Stopping hands the agents it holds on to the ones on their way out. Run over an empty - list it would hand nothing on and let go of the ones already there, so the press that - does not wait for the flow would find no conversation left to close. + Stopping hands the run on to the one on its way out. Told again with nothing running it + would hand nothing on and let go of the one already there, so the press that does not + wait for the flow would find no conversation left to close. """ - from hmz.coganchor.agents import ClaudeCodeAgent, ClaudeCodeAgentConfig - app = Humanize() async with app.run_test() as driver: - app._stopping = [ - ClaudeCodeAgent(ClaudeCodeAgentConfig(model="claude-opus-5", effort="high")) - ] + app._stopping = Holding() held = app._stopping await _typed(driver, "/stop") @@ -217,18 +198,15 @@ async def test_the_press_after_a_typed_stop_does_not_close_the_interface( written(workspace, "flow", FLOW) app = Humanize() async with app.run_test() as driver: - app._flow_named, app._models = "flow", [Runs("claude/m:high")] + set_up(app, "flow") await _typed(driver, "start") - await until( - lambda: bool(app._agents and any(agent.sessions for agent in app._agents)), - driver, - ) + await until(lambda: any(agent.sessions for agent in app._agents), driver) await driver.press("ctrl+c") await driver.pause() assert app._presses == 1 await _typed(driver, "/stop") - await until(lambda: not app._agents and not app._stopping, driver) + await until(lambda: app._run is None and app._stopping is None, driver) await driver.press("ctrl+c") await driver.pause() @@ -245,16 +223,12 @@ async def test_the_keys_name_what_the_press_after_a_typed_stop_does() -> None: would offer to leave on a key that closes the conversations still open under their turns. Counting from nothing again is what keeps the row true as well as the key. """ - from hmz.coganchor.agents import ClaudeCodeAgent, ClaudeCodeAgentConfig - app = Humanize() async with app.run_test() as driver: await driver.press("ctrl+c") # the press that asks, and is then typed past await driver.pause() assert app._presses == 1 - app._stopping = [ - ClaudeCodeAgent(ClaudeCodeAgentConfig(model="claude-opus-5", effort="high")) - ] + app._stopping = Holding() await _typed(driver, "/stop") await until(lambda: "already stopping" in transcript(app), driver) diff --git a/tests/recording.py b/tests/recording.py index 24ac3197..9e7db5fd 100644 --- a/tests/recording.py +++ b/tests/recording.py @@ -1,11 +1,15 @@ -"""One run written down, and what is read back off it -- shared by four test modules. +"""One run written down, and what is read back off it -- shared by the runtime's test modules. An epic and an export are one subject read twice: a run opens a session, the session's log is linked into the run's directory, and a bundle is that directory with every link followed. Both -subjects split across tiers -- what a stand-in agent can show, and what only a turn taken as a +subjects split across tiers -- what a stand-in CLI can show, and what only a turn taken as a named account can, which needs a kernel that will hand over a tracee -- so the flow that opens -one session, the stand-in whose logs humanize knows where to find, and the three readers below -are wanted by `tests/{integration,system}/runtime/test_{epics,export}.py`. +one session, the stand-in whose logs humanize knows where to find, and the readers below are +wanted by `tests/{integration,system}/runtime/test_{epics,export}.py`. + +The stand-in is :data:`tests.flows.standins.CLAUDE` -- a `claude` on PATH that speaks the real +CLI's stream JSON, takes the session id it is handed, and keeps each conversation where Claude +Code keeps one, which is where humanize links it from. Written once here rather than four times there because a bundle's layout is the thing these read: a copy of `held` in a file CI never runs is one that goes on asserting last year's tar @@ -19,52 +23,71 @@ from __future__ import annotations import json +import re import tarfile from typing import TYPE_CHECKING, Any from hmz.runtime.exporting import MANIFEST -from tests.stubs import ShellAgent +from tests.flows import standins if TYPE_CHECKING: from pathlib import Path import pytest -#: A flow that drives one agent, and says its session is called what the log is named after. -ONE = """ -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +#: A flow that drives one agent, `builder`, through one session in the workspace. +ONE = '''"""Opens one session, and says the task in it.""" +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - agents[0].new()("echo the-session") -""" +class Agents(AgentCollection): + builder: Agent + + +class Envs(EnvCollection): + here: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def one(task, *, agents, envs, params, ctx): + session = await agents["builder"].spawn(env=envs["here"]) + return await agents["builder"].run(task, session=session) +''' + +#: What the stand-in is driven as, after `builder=`. +AGENT = "claude/claude-haiku-4-5:low" -class ClaudeAgent(ShellAgent): - """A stand-in for a backend humanize knows where the logs of are.""" +#: What is run on it: a turn the stand-in answers with the word. +TASK = "Reply with the single word: done" -def claude_home( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch, said: str = "{}" -) -> Path: - """Points Claude Code's home somewhere temporary, with one session already logged. +def standing_in(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + """Puts the stand-in `claude` on PATH, with a home and a config directory of its own. Args: - tmp_path: The test's own directory, which the home goes under. - monkeypatch: What sets the variable, and puts it back afterwards. - said: The one line the logged session holds. + tmp_path: The test's own directory, which all of it goes under. + monkeypatch: What sets the variables, and puts them back afterwards. Returns: - The log that session was written to. + Where it keeps its conversations: `projects//.jsonl`. """ - where = tmp_path / "claude-home" - monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(where)) - log = where / "projects" / "-tmp-project" / "the-session.jsonl" - log.parent.mkdir(parents=True) - log.write_text(f"{said}\n", encoding="utf-8") - return log + standins.install(tmp_path / "bin", "claude", standins.CLAUDE) + monkeypatch.setenv("PATH", standins.path_with(tmp_path / "bin")) + monkeypatch.setenv("HOME", str(tmp_path / "home")) + config = tmp_path / "claude-home" + monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(config)) + return config + + +def logged(config: Path, workspace: Path, session: str) -> Path: + """Where the stand-in keeps one conversation, as Claude Code keeps one.""" + return ( + config + / "projects" + / re.sub(r"[^a-zA-Z0-9]", "-", str(workspace)) + / (f"{session}.jsonl") + ) def held(at: Path) -> dict[str, str]: diff --git a/tests/system/machines/test_isolation.py b/tests/system/machines/test_isolation.py index f3950199..44080419 100644 --- a/tests/system/machines/test_isolation.py +++ b/tests/system/machines/test_isolation.py @@ -1,11 +1,10 @@ -"""A container really started, and the turns and flows that then run inside it. +"""A container really started, and the turns that then run inside it. Every test here brings up a container of a pulled image and takes it down again, because the things being checked are the ones only a real one can answer: that the workspace is the directory this machine already had rather than a copy, that a turn ran somewhere that is not -this host, that a flow naming an image lands its agent there, and that the container is gone -when the run that asked for it ends. A stand-in cannot say any of that -- it would only repeat -what the test told it. +this host, and that the container is gone when whatever asked for it is done with it. A +stand-in cannot say any of that -- it would only repeat what the test told it. That means a docker daemon, a pulled `python:3.12-slim`, and a user who may talk to the socket. CI is not given those, so CI does not run this file; the half that needs none of them @@ -20,18 +19,17 @@ import subprocess from typing import TYPE_CHECKING -import pytest - from hmz.coganchor import check from hmz.coganchor.agents import AgentConfig from hmz.coganchor.machines import DockerConfig -from hmz.runtime.runner import Runner from tests.machines.fixtures import IMAGE -from tests.stubs import ShellAgent, written +from tests.stubs import ShellAgent if TYPE_CHECKING: from pathlib import Path + import pytest + def _inspect(container: str, field: str) -> str: """What docker says about one container, or "" when there is no such container.""" @@ -85,51 +83,6 @@ def test_a_turn_runs_in_the_container_and_leaves_its_work_in_the_workspace( assert (tmp_path / "stamp.txt").read_text().strip() == answer -#: A flow that says one of its agents works in a container of an image it names, and has that -#: agent leave a mark where a person could find it afterwards. -ISOLATING = f''' -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Isolated -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one this drives, in a container of its own.""" - - tester: Annotated[AgentBase, Isolated("{IMAGE}")] - - -@flow -def run(agents: Agents, task: str) -> None: - agents.tester("hostname > stamp.txt; cat /etc/os-release > which.txt") -''' - - -def test_a_flow_that_isolates_an_agent_runs_it_in_the_image_it_named( - daemon: None, tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Isolated mode, end to end: the flow names the image and nobody configures anything. - - The container is started for the agent, the project directory is mounted into it at the - path it already has, and the turn runs there through coganchor -- so the work is this - machine's file in this machine's directory, and the tools that did it were the image's. - """ - monkeypatch.chdir(tmp_path) - (tmp_path / "flow.py").write_text(ISOLATING) - agent = ShellAgent(AgentConfig(model="m", effort="high")) - - Runner(tmp_path / "flow.py", [agent]).run("go") - - # Nobody said where it works, and it works in a container of the image the flow named. - machine = agent.config.machine - assert isinstance(machine, DockerConfig) - assert machine.image == IMAGE - # The work is here, and it was done there. - assert (tmp_path / "stamp.txt").read_text().strip() != socket.gethostname() - assert "Debian" in (tmp_path / "which.txt").read_text() - - def test_a_session_may_be_opened_at_a_directory_on_the_machine_it_lands_on( daemon: None, tmp_path: Path ) -> None: @@ -153,132 +106,6 @@ def test_a_session_may_be_opened_at_a_directory_on_the_machine_it_lands_on( assert (tmp_path / "packages" / "one" / "where.txt").exists() -#: A flow with two agents and nothing said about where either works, which is what a run put -#: in a container from outside is: the flow did not ask for one, and every agent lands there. -CONTAINED = """ -from typing import NamedTuple - -from hmz._legacy_flows import Agent, container, flow - - -class Agents(NamedTuple): - builder: Agent - reviewer: Agent - - -@flow -def run(agents: Agents, task: str) -> None: - agents.builder("hostname > builder.txt") - agents.reviewer("hostname > reviewer.txt") - held = container() - held.write_text("from-the-flow.txt", held.run(["hostname"]).output) -""" - - -def test_a_run_may_be_put_in_one_container_and_every_agent_lands_there( - daemon: None, tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """Which is the convenience: said once from outside rather than agent by agent inside. - - One container for the run rather than one apiece, so that what one agent writes is what - the next one reads -- and the flow's own code reaches the same place, which is the half a - mounted workspace does not answer for. - """ - monkeypatch.chdir(tmp_path) - (tmp_path / "flow.py").write_text(CONTAINED) - agents = [ - ShellAgent(AgentConfig(model="m", effort="high")), - ShellAgent(AgentConfig(model="m", effort="high")), - ] - - Runner(tmp_path / "flow.py", agents, container=IMAGE).run("go") - - # Both turns ran on the machine, and on the same one. - said = [ - (tmp_path / one).read_text().strip() - for one in ("builder.txt", "reviewer.txt", "from-the-flow.txt") - ] - assert said[0] == said[1] == said[2] - assert said[0] != socket.gethostname() - # And it is taken down when the run ends, whichever way it ends. - assert _inspect(said[0], "{{.State.Running}}") == "" - - -#: The flow a contained run is started at, which calls one whose place says nothing about -#: where its agent works -- which is what every flow written before this said. -CALLING = '''"""Calls a flow that says nothing about where its agent works.""" - -from hmz._legacy_flows import Agent, flow, load - - -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - agent.new()("hostname > calling.txt") - load("called")(agents, task) -''' - -#: And the one it calls, which has its agent leave a mark saying where the turn ran. -CALLED = '''"""The one that is called, driving the agents it was handed.""" - -from hmz._legacy_flows import Agent, flow - - -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - agent.new()("hostname > called.txt") -''' - - -def test_a_run_in_a_container_may_call_a_flow_that_says_where_nobody_works( - daemon: None, tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A place that says nothing refuses a machine somebody chose, and this is not one. - - The run's container was said once from outside, about every agent, by whoever started - the run -- so a called flow reading it as a machine anybody reached for would refuse - every flow written before there was such a thing, and name itself in the refusal. - """ - monkeypatch.chdir(tmp_path) - written(tmp_path / ".humanize/flows", "calling", CALLING) - written(tmp_path / ".humanize/flows", "called", CALLED) - agent = ShellAgent(AgentConfig(model="m", effort="high")) - - Runner("calling", [agent], container=IMAGE).run("go") - - # The call went through, and its turn ran in the run's container rather than here -- - # the same one the flow that called it was working in, which is the point of one - # container for the run: what the caller wrote is what the called flow reads. - called = (tmp_path / "called.txt").read_text().strip() - assert called != socket.gethostname() - assert called == (tmp_path / "calling.txt").read_text().strip() - - -def test_a_second_run_in_a_container_is_refused_rather_than_handed_the_first( - daemon: None, tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """The container is the process's rather than the run's, so there is one to be in. - - Two started at once would be two runs sharing a workspace neither was told about, and - the second of them reading the first's container back as its own. - """ - from hmz.runtime.flowing.driving import contained - - monkeypatch.chdir(tmp_path) - with contained(IMAGE) as where_: - assert where_ is not None - with ( - pytest.raises(RuntimeError, match="in a container already"), - contained(IMAGE), - ): - raise AssertionError # never reached: the refusal is at the block's opening - # And a run on this machine is not a second run in a container: it starts nothing, - # so there is nothing for it to be reaching for. - with contained("") as none: - assert none is None - - def test_the_flow_reads_writes_and_runs_on_the_machine_the_run_lands_on( daemon: None, tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/tests/system/runtime/test_epics.py b/tests/system/runtime/test_epics.py index a37c77fb..e9a46e48 100644 --- a/tests/system/runtime/test_epics.py +++ b/tests/system/runtime/test_epics.py @@ -1,13 +1,13 @@ """The half of one run's record that only a machine which can supervise a turn can check. -The rest of it is `tests/integration/runtime/test_epics.py`, which drives a stand-in agent and -asks what was written down. Here is the one thing that stand-in cannot show: a turn taken as a +The rest of it is `tests/integration/runtime/test_epics.py`, which drives a stand-in CLI and +asks what was written down. Here is the one thing that cannot be shown there: a turn taken as a named account is a supervised turn -- the paths it reads are answered by others, which is a seccomp filter and a ptrace supervisor -- so what an epic says about *which* account ran a session can only be checked where the kernel will hand over a tracee. CI is not promised one, which is why this is a tier of its own rather than a skip inside the other file. -What the two halves share -- a flow that opens one session, and a stand-in for a backend whose +What the two halves share -- a flow that opens one session, and the stand-in `claude` whose logs humanize knows where to find -- is in `tests/recording.py`, where the exporter's two halves reach for it as well. """ @@ -16,10 +16,9 @@ from typing import TYPE_CHECKING -from hmz.coganchor.agents import AgentConfig from hmz.runtime.epic import epics, read, sessions from hmz.runtime.runner import Runner -from tests.recording import ONE, ClaudeAgent +from tests.recording import ONE, TASK, standing_in from tests.stubs import written from tests.supervising import traced @@ -36,21 +35,23 @@ def test_a_session_says_which_account_took_its_turns( """Two agents of one CLI are two accounts, and the backend's log says neither.""" from hmz.coganchor import providers + standing_in(tmp_path, monkeypatch) monkeypatch.chdir(tmp_path) written(tmp_path, "flow", ONE) providers.add("claude", "work", "key", {"ANTHROPIC_API_KEY": "sk-nothing"}) - agent = ClaudeAgent( - AgentConfig(model="m", effort="high", provider="work"), name="builder" - ) - Runner(tmp_path / "flow", [agent]).run("go") + Runner( + tmp_path / "flow", + agents={"builder": "claude@work/claude-haiku-4-5:low"}, + budget={"cost": 1}, + ).run(TASK) (epic,) = epics() (one,) = sessions(epic) assert one.provider == "work" - assert one.name == "builder-claude@work-the-session" + assert one.name == f"builder-claude@work-{one.ident}" # And what it was configured with is what the run says it was driven by. ran = read(epic) assert ran is not None assert ran.agents[0].provider == "work" - assert ran.agents[0].spec == "claude@work/m:high" + assert ran.agents[0].spec == "claude@work/claude-haiku-4-5:low" diff --git a/tests/system/runtime/test_exec.py b/tests/system/runtime/test_exec.py new file mode 100644 index 00000000..fed5c48f --- /dev/null +++ b/tests/system/runtime/test_exec.py @@ -0,0 +1,176 @@ +"""`hmz exec` driving a real coding agent CLI, as the account this machine is signed in as. + +The stand-in half is `tests/integration/cli/test_run_command.py`, which holds the line, its +refusals and the flow to a stand-in `claude`. What only the real thing can show is that a line +naming a real harness gets a turn out of it: a flow of this project's own with a role, a +workspace and a param, run under a budget, and `chat`, which is run with none. Claude Code on +its cheapest model, since that is one prompt a test; a machine without it installed, or signed +out of it, skips saying so. + +These cost tokens and need network access, so they only run with ``pytest --run-agents``. +""" + +from __future__ import annotations + +import json +import shutil +from typing import TYPE_CHECKING + +import pytest + +from hmz.cli import main +from hmz.flows import HarnessError +from hmz.runtime.epic import epics, read, sessions +from tests.stubs import written + +if TYPE_CHECKING: + from pathlib import Path + +pytestmark = [pytest.mark.agent, pytest.mark.timeout(600)] + +#: The cheapest model Claude Code takes, at the least effort it takes. +WORKER = "claude/claude-haiku-4-5-20251001:low" + +#: A flow of this project's own: one role, the workspace it is started in, and a param. +DEMO = '''"""Asks for a word a round, and writes down what came back.""" + +from hmz.flows import ( + Agent, + AgentCollection, + EnvCollection, + FilesEnvMixin, + FlowParams, + LocalEnv, + flow, +) + + +class Workspace(LocalEnv, FilesEnvMixin): ... + + +class Agents(AgentCollection): + worker: Agent + + +class Envs(EnvCollection): + workspace: Workspace + + +class Params(FlowParams): + rounds: int = 1 + + +@flow(agents=Agents, envs=Envs, params=Params) +async def demo(task, *, agents, envs, params, ctx): + worker, here = agents["worker"], envs["workspace"] + session = await worker.spawn(env=here) + said = [] + for n in range(params.rounds): + said.append( + await worker.run( + f"Reply with exactly one word, the word round{n}, and nothing else.", + session=session, + ) + ) + await here.write("demo.json", repr(said).encode()) + return said +''' + + +@pytest.fixture +def project(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + """A project of its own, with the demo flow in it; skipped where there is no `claude`.""" + if shutil.which("claude") is None: + pytest.skip("claude is not installed here") + written(tmp_path / ".humanize" / "flows", "demo", DEMO) + monkeypatch.chdir(tmp_path) + return tmp_path + + +def _exec(*argv: str) -> int: + """One `hmz exec` line, skipped where the account refuses rather than the run failing.""" + try: + return main(["exec", *argv]) + except HarnessError as refused: + pytest.skip(f"claude would not take the turn here: {refused}") + + +def test_a_flow_of_your_own_runs_on_a_real_cli(project: Path) -> None: + assert ( + _exec( + "-f", + "demo", + "-a", + f"worker={WORKER}", + "-p", + "rounds=2", + "-b", + "cost=0.5", + "two words", + ) + == 0 + ) + + said = (project / "demo.json").read_text() + assert "round0" in said + assert "round1" in said + (epic,) = epics() + ran = read(epic) + assert ran is not None + assert ran.how == "done" + assert ran.params == {"rounds": 2} + # One session, named for its role, and the output tokens the run spent written down. + (one,) = sessions(epic) + assert one.agent == "worker" + usage = [ + json.loads(line) + for line in (epic / "epic.jsonl").read_text().splitlines() + if '"usage"' in line + ] + assert usage[-1]["output_tokens"] > 0 + + +def test_chat_runs_on_a_real_cli_with_no_budget_and_ends_after_one_turn( + project: Path, capsys: pytest.CaptureFixture[str] +) -> None: + assert ( + _exec( + "-f", + "chat", + "-a", + f"assistant={WORKER}", + "Reply with exactly one word, the word pineapple, and nothing else.", + ) + == 0 + ) + + assert "pineapple" in capsys.readouterr().out.lower() + (epic,) = epics() + ran = read(epic) + assert ran is not None + assert ran.budget is not None + assert ran.budget["cost"] == "Infinity" + + +def test_a_run_written_for_a_program_is_one_object_a_line( + project: Path, capsys: pytest.CaptureFixture[str] +) -> None: + assert ( + _exec( + "-f", + "demo", + "-a", + f"worker={WORKER}", + "-b", + "cost=0.5", + "--json", + "one word", + ) + == 0 + ) + + lines = capsys.readouterr().out.splitlines() + assert lines + objects = [json.loads(line) for line in lines] + assert {one["agent"] for one in objects} == {"worker"} + assert any(one["kind"] == "result" for one in objects) diff --git a/tests/system/runtime/test_export.py b/tests/system/runtime/test_export.py index 8a04a6ab..b0c26084 100644 --- a/tests/system/runtime/test_export.py +++ b/tests/system/runtime/test_export.py @@ -2,7 +2,7 @@ The rest of the exporter is `tests/integration/runtime/test_export.py`: what a bundle holds, what its manifest says, where it lands and how big it came out, all of it against a stand-in -agent. Here is the promise that a run taken as a named account carries none of that account's +CLI. Here is the promise that a run taken as a named account carries none of that account's key -- and a turn under an account is a turn whose reads are answered by others, which is a seccomp filter and a ptrace supervisor. CI is not promised a kernel that will hand one over, so this is a tier of its own rather than a skip inside the other file: a redaction test that @@ -19,11 +19,10 @@ from typing import TYPE_CHECKING from hmz.coganchor import providers -from hmz.coganchor.agents import AgentConfig from hmz.runtime.epic import epics from hmz.runtime.exporting import REDACTED, bundle from hmz.runtime.runner import Runner -from tests.recording import ONE, ClaudeAgent, claude_home, held, manifest +from tests.recording import ONE, held, manifest, standing_in from tests.stubs import written from tests.supervising import traced @@ -37,7 +36,12 @@ def test_no_account_variable_rides_along( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: - """An export is the user's to send. The key it ran on is nobody's.""" + """An export is the user's to send. The key it ran on is nobody's. + + The key is in the log this time: the turn was asked to say it, and the stand-in keeps + what it was asked in the conversation's own log -- which the bundle carries, struck. + """ + standing_in(tmp_path, monkeypatch) monkeypatch.chdir(tmp_path) written(tmp_path, "flow", ONE) providers.add( @@ -46,16 +50,12 @@ def test_no_account_variable_rides_along( "key", {"ANTHROPIC_AUTH_TOKEN": "hunter2-hunter2-hunter2"}, ) - claude_home( - tmp_path, - monkeypatch, - '{"headers":{"x-api-key":"hunter2-hunter2-hunter2"}}', - ) - agent = ClaudeAgent( - AgentConfig(model="m", effort="high", provider="work"), name="builder" - ) - Runner(tmp_path / "flow", [agent]).run("go") + Runner( + tmp_path / "flow", + agents={"builder": "claude@work/claude-haiku-4-5:low"}, + budget={"cost": 1}, + ).run("Reply with the single word: hunter2-hunter2-hunter2") (epic,) = epics() inside = held(bundle(epic, tmp_path / "out.tar.gz")[0]) diff --git a/tests/system/runtime/test_together.py b/tests/system/runtime/test_together.py index 5fe6c5c4..217e9c12 100644 --- a/tests/system/runtime/test_together.py +++ b/tests/system/runtime/test_together.py @@ -107,7 +107,6 @@ def test_one_flow_runs_two_agents_of_one_cli_as_two_accounts( import json as reading from hmz.coganchor import providers - from hmz.coganchor.agents import ClaudeCodeAgent, ClaudeCodeAgentConfig from hmz.runtime.runner import Runner binaries = tmp_path / "bin" @@ -135,23 +134,35 @@ def test_one_flow_runs_two_agents_of_one_cli_as_two_accounts( import json from pathlib import Path -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowParams, LocalEnv, flow -@flow -def run(agents: tuple[AgentBase, AgentBase], task: str) -> None: - Path("said.json").write_text(json.dumps([agent(task) for agent in agents])) +class Agents(AgentCollection): + subscription: Agent + gateway: Agent + + +class Envs(EnvCollection): + here: LocalEnv + + +@flow(agents=Agents, envs=Envs, params=FlowParams) +async def run(task, *, agents, envs, params, ctx): + said = [] + for role in ("subscription", "gateway"): + session = await agents[role].spawn(env=envs["here"]) + said.append(await agents[role].run(task, session=session)) + Path("said.json").write_text(json.dumps(said)) """, ) - agents = [ - ClaudeCodeAgent( - ClaudeCodeAgentConfig(model="m", effort="high", provider=named), name=named - ) - for named in ("subscription", "gateway") - ] - Runner(workspace / "flow", agents).run("who are you") + Runner( + workspace / "flow", + agents={ + named: f"claude@{named}/m:high" for named in ("subscription", "gateway") + }, + budget={"cost": 1}, + ).run("who are you") assert reading.loads((workspace / "said.json").read_text()) == [ '"subscription"', diff --git a/tests/system/tui/test_app.py b/tests/system/tui/test_app.py index 5197dbf7..b4a857cf 100644 --- a/tests/system/tui/test_app.py +++ b/tests/system/tui/test_app.py @@ -46,8 +46,8 @@ async def test_deepseek_chat_explains_a_missing_api_key_instead_of_staying_blank ) -> None: monkeypatch.delenv("DEEPSEEK_API_KEY", raising=False) monkeypatch.setenv("DSH_HOME", str(tmp_path / "dsh-home")) - app = Humanize(agents=[Runs("dsh/deepseek-v4-flash:high")]) - assert app._models == [Runs("dsh/deepseek-v4-flash:high")] + app = Humanize(agents={"assistant": Runs("dsh/deepseek-v4-flash:high")}) + assert app._models == {"assistant": Runs("dsh/deepseek-v4-flash:high")} async with app.run_test() as driver: await driver.press(*"hello") diff --git a/tests/tui/fixtures.py b/tests/tui/fixtures.py index 091ebcf0..fb4e5edc 100644 --- a/tests/tui/fixtures.py +++ b/tests/tui/fixtures.py @@ -46,11 +46,14 @@ from hmz.tui.selecting import Transcript if TYPE_CHECKING: - from collections.abc import Callable + from collections.abc import Callable, Mapping from pathlib import Path from textual.pilot import Pilot + from hmz.coganchor.agents import AgentBase + from hmz.flows import Budget + from hmz.runtime.kept import Runs from hmz.tui import Humanize #: How long anything here waits for the interface to catch up before giving up on it. @@ -139,3 +142,114 @@ def transcript(app: Humanize) -> str: Read while the interface is still up: its widgets go with it when it exits. """ return app.query_one("#transcript", Transcript).text + + +#: A flow of one agent role, `coder`, working in the workspace it was started in: one turn on +#: the task, and what that turn answered written beside it to `said.txt`. Written against the +#: flow API, as every flow a test here runs is, and run through the runtime on whatever `-a` +#: the interface is set up with. +ONE = """ +from pathlib import Path + +from hmz.flows import Agent, AgentCollection, EnvCollection, FlowContext, FlowParams +from hmz.flows import LocalEnv, flow + + +class Agents(AgentCollection): + coder: Agent + + +class Envs(EnvCollection): + workspace: LocalEnv + + +class Params(FlowParams): + pass + + +@flow(agents=Agents, envs=Envs, params=Params, name="flow") +async def run(task: str, *, agents: Agents, envs: Envs, params: Params, ctx: FlowContext): + coder = agents["coder"] + session = await coder.spawn(env=envs["workspace"]) + Path("said.txt").write_text(await coder.run(task, session=session) + "\\n") +""" + + +def set_up( + app: Humanize, + flow: str, + agents: Mapping[str, Runs] | None = None, + *, + budget: Budget | None = None, +) -> None: + """Sets the interface up to run one flow, as saving the flow menu would. + + Args: + app: The interface. + flow: The flow, by the name it is offered under or a path. + agents: What each agent role runs, by role; `claude/m:high` for `coder` where None. + budget: What a run of it may spend; a dollar where None. + """ + from hmz.flows import Budget + from hmz.runtime.kept import Runs + from hmz.tui.pick import declared_of + + app._flow_named = flow + app._declared = declared_of(flow) + app._models = ( + dict(agents) if agents is not None else {"coder": Runs("claude/m:high")} + ) + app._budget = budget if budget is not None else Budget(cost=1) + + +class Holding: + """A run the interface is holding that runs nothing: for a test about that state alone. + + Attributes: + stopped: Whether it was told to stop. + closed: Whether it was closed. + """ + + flow = "flow" + + def __init__(self) -> None: + from hmz.flows import Budget, Usage + + self.budget = Budget(cost=1) + self.usage = Usage() + self.stopped = False + self.closed = False + + def watch(self, listener: object) -> None: + """Hears nothing, there being nothing to hear.""" + + def opened(self, callback: object) -> None: + """Opens nothing, and so tells nothing.""" + + def run(self) -> None: + """Runs nothing.""" + + def stop(self) -> None: + """Writes down that it was told to.""" + self.stopped = True + + def close(self) -> None: + """Likewise.""" + self.closed = True + + +def holding(app: Humanize, *agents: AgentBase) -> Holding: + """Puts the interface in the state of holding a running flow, with these agents in it. + + Args: + app: The interface. + agents: The agents behind the sessions the run has opened. + + Returns: + The run it is holding. + """ + run = Holding() + app._run = run + app._agents = list(agents) + app._ran = app._agents + return run diff --git a/tests/unit/agents/test_web_search.py b/tests/unit/agents/test_web_search.py index 76e16c61..ce3c45cd 100644 --- a/tests/unit/agents/test_web_search.py +++ b/tests/unit/agents/test_web_search.py @@ -11,11 +11,9 @@ import json from dataclasses import replace -from typing import TYPE_CHECKING import pytest -from hmz._legacy_flows import NotAFlow from hmz.coganchor import backends from hmz.coganchor.agents import ( ClaudeCodeAgent, @@ -32,10 +30,6 @@ QwenCodeAgent, QwenCodeAgentConfig, ) -from hmz.runtime.runner import Runner - -if TYPE_CHECKING: - from pathlib import Path # : The backends that can be told, and the ones that cannot. Read off `hmz.coganchor.backends` here # as : everything else reads it, so a backend that gains a way of being told is a backend this : @@ -52,23 +46,6 @@ "zcode", ) -#: A flow whose one agent reads this repository and nothing else, which is a thing about the -#: work: it says so where it declares the place, and nobody outside it may say otherwise. -SEARCHLESS = '''"""A flow whose answers have to be the same tomorrow.""" - -from typing import Annotated - -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow - - -@flow -def run( - agents: tuple[Annotated[AgentBase, AgentDefaults(web_search=False)]], task: str -) -> None: - agents[0](task) -''' - def test_an_agent_nobody_has_been_asked_about_is_told_neither_way() -> None: """Whether it reads the internet is then the CLI's own answer rather than humanize's.""" @@ -332,26 +309,6 @@ def test_the_flow_says_it_and_a_line_that_says_it_is_refused() -> None: assert backends.read("claude/m:high")[1].name == "claude" -def test_what_a_flow_declares_reaches_the_agent(tmp_path: Path) -> None: - """Settled onto it before the first turn, over whatever it was made with.""" - where = tmp_path / "quiet.py" - where.write_text(SEARCHLESS) - agent = ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="high")) - - Runner(str(where), [agent]) - - assert agent.config.web_search is False - - -def test_a_backend_that_cannot_be_told_cannot_fill_such_a_place(tmp_path: Path) -> None: - """Refused where the run is set up, rather than searching on under a flow that says not.""" - where = tmp_path / "quiet.py" - where.write_text(SEARCHLESS) - - with pytest.raises(NotAFlow, match="no way of being told not to search the web"): - Runner(str(where), [PiAgent(PiAgentConfig(model="m", effort="high"))]) - - def test_a_backend_that_cannot_be_told_takes_the_silence() -> None: """Off is what it refuses. Nothing said is not an off. diff --git a/tests/unit/backends/test_catalogue.py b/tests/unit/backends/test_catalogue.py index b25a185c..ec4761ba 100644 --- a/tests/unit/backends/test_catalogue.py +++ b/tests/unit/backends/test_catalogue.py @@ -1,38 +1,21 @@ -"""The capability catalogue, held to saying only what this installation actually serves. - -Honesty tests: every name the catalogue uses is a real moment, a real backend or a real -member of the interfaces a flow is written against, and every backend set is exactly what -the live driver classes declare or what the facts in `hmz.coganchor.backends` say. The catalogue is -what a compiler steers by, and a capability it invented -- or one that drifted from the -drivers -- is a generated flow that asks for what nothing serves. - -The vocabulary those names are drawn from is covered here too: which backends steer a turn -already running, and which capability names a backend's own facts come to. What a capability -does once it has been asked for is covered where it is driven -- steering a real CLI in -`tests/system/agents` -- and not here. +"""The facts about each backend, held to what its driver actually does. + +Honesty tests: every backend set here is exactly what the live driver classes declare or what +the facts in `hmz.coganchor.backends` say -- which backends steer a turn already running, which +names a backend's own facts come to, what each counts and which rungs each takes. A fact that +drifted from the driver is a capability humanize would offer and nothing serves. What a +capability does once it has been asked for is covered where it is driven -- steering a real CLI +in `tests/system/agents` -- and not here. """ from __future__ import annotations -import dataclasses import inspect -import re import sys from typing import TYPE_CHECKING -from hmz._legacy_flows import Agent, Person, Session -from hmz.coganchor import places -from hmz.coganchor.agents import DRIVEN, EVERYWHERE, KINDS, PERMISSIONS, Moment, rung -from hmz.coganchor.agents.config import UNSAID, AgentConfig -from hmz.coganchor.backends import PROFILES, Bundled, Hooked, Profile, named -from hmz.runtime.flowing.checking import ( - OF_AGENT, - WHERE, - briefed, - catalogue, - offered, - surface, -) +from hmz.coganchor.agents import DRIVEN, KINDS, PERMISSIONS +from hmz.coganchor.backends import PROFILES, Bundled, Hooked, Profile if TYPE_CHECKING: from hmz.coganchor.agents.base import SessionBase @@ -60,90 +43,6 @@ def _sessions() -> dict[str, type[SessionBase]]: return held -def test_every_conditional_moment_is_real_and_exactly_whose_drivers_say() -> None: - told = {one.name: one for one in catalogue() if one.name.startswith("moment:")} - outside = {one for one in Moment if one not in EVERYWHERE} - assert set(told) == {f"moment:{one.value}" for one in outside} - for moment in outside: - assert told[f"moment:{moment.value}"].backends == frozenset( - name for name, (cls, _) in DRIVEN.items() if moment in cls.moments - ) - - -def test_the_backend_facts_are_the_drivers_own() -> None: - told = {one.name: one.backends for one in catalogue()} - assert told["pursue"] == frozenset( - name for name, (cls, _) in DRIVEN.items() if cls.pursues - ) - assert told["goal"] == told["pursue"] - # The two facts a session carries, checked against the backends known to carry them: - # the sets themselves are read off the session classes, so what is pinned here is that - # the reading reaches them at all. - assert {"claude", "codex"} <= told["shape"] - assert "claude" in told["tools"] - for one in catalogue(): - assert one.backends <= set(DRIVEN), one.name - - -def test_every_ask_the_catalogue_spells_is_on_the_interfaces() -> None: - """The primitives are described in code, and the code has to be the real interface.""" - asks = surface(Agent) | surface(Session) | surface(Person) - anchored = { - "turns": "batch", - "sessions": "new", - "budgets": "spent", - "hooks": "hooks", - "board": "board", - "clone": "clone", - "skills": "loads", - "pursue": "pursue", - "tools": "offers", - "steer": "interject", - "fork": "fork", - } - said = {one.name: one.said for one in catalogue()} - for name, member in anchored.items(): - assert member in asks - assert member in said[name], name - # And the ones whose anchor is the vocabulary hmz._legacy_flows hands through. - offers = offered() - for name, word in { - "subflows": "load", - "person": "Person", - "state": "flow", - "hooks": "Moment", - "goal": "Goal", - }.items(): - assert word in offers - assert word in said[name], name - - -def test_the_moments_every_backend_reaches_are_everywhere() -> None: - (moments,) = (one for one in catalogue() if one.name == "moments") - assert moments.backends == frozenset() - for one in EVERYWHERE: - assert f"Moment.{one.name}" in moments.said - - -def test_the_briefing_mentions_every_capability_and_its_backends() -> None: - page = briefed() - for one in catalogue(): - assert f"- {one.name}" in page - for backend in one.backends: - assert backend in page - # The split the compiler steers by: what needs declaring is under the second heading, - # and what a *machine* has to come to is under a third of its own. Three rather than two - # because a flow writes the vocabulary in two places -- `Needs(...)` and `Needs(where=...)` - # -- and a page that ran them together was teaching the one mistake nothing used to catch. - assert "Every backend -- Needs(...) beside the place:" in page - assert "Only some backends" in page - assert "Where an agent's turns land -- Needs(where=(...))" in page - assert page.index("- turns:") < page.index("Only some backends") - assert page.index("Only some backends") < page.index("- pursue") - assert page.index("- pursue") < page.index("Where an agent's turns land") - assert page.index("Where an agent's turns land") < page.index("- isolated:") - - def test_a_backend_that_steers_a_running_turn_says_so_and_one_that_cannot_says_so() -> ( None ): @@ -156,67 +55,6 @@ def test_a_backend_that_steers_a_running_turn_says_so_and_one_that_cannot_says_s assert one.steers is (name in _STEERING), name -def test_the_catalogue_says_which_backends_narrate_a_reach_as_it_happens() -> None: - """One, and it is the one whose driver asks its CLI for the fragments.""" - sessions = _sessions() - told = {one.name: one.backends for one in catalogue()} - - assert told["narrate"] == frozenset( - name for name, one in sessions.items() if one.narrates - ) - assert told["narrate"] == {"claude"} - - -def test_one_setting_is_one_capability_name_and_no_second_one() -> None: - """A field only some of these configs carry is asked for as `settings:` alone. - - The catalogue used to mint a word of its own for some of them -- `trust`, `title`, - `features` -- beside the name the derived block gives every such field, so Cursor's trust - flag answered to `trust` and to `settings:trust` both. That is one fact written in two - places: the hand-written word goes on promising what a renamed field no longer serves, - and a generated flow asks under whichever of the two it happened to read. - - What makes a hand-written name a duplicate rather than a capability of its own is that it - says nothing the field's own presence does not. `narrate` is the counter-case and is why - this is not simply a ban on naming a field: it asks whether a session can be told at all, - which is a question with a different answer -- two backends carry `partial_messages` and - one of them narrates. So a name is a duplicate here when it serves exactly the backends - whose config carries the field *and* names that field, and only the derived one may. - """ - common = {one.name for one in dataclasses.fields(AgentConfig)} - carriers: dict[str, set[str]] = {} - for backend, (_, config) in DRIVEN.items(): - for one in dataclasses.fields(config): - if one.name not in common: - carriers.setdefault(one.name, set()).add(backend) - held = catalogue() - - assert carriers - for field, backends in carriers.items(): - naming = [ - one.name - for one in held - if one.backends == frozenset(backends) - and re.search(rf"\b{field}\b", one.said) - ] - assert naming == [f"settings:{field}"], field - # `tier:fast` is not one of these and must not be read as one: `service_tier` is a field - # of the common config, which every backend carries and the derived block therefore - # leaves out, so the hand-written name is the only word there has ever been for it. - assert "service_tier" in common - assert "tier:fast" in {one.name for one in held} - - -def test_the_catalogue_says_which_backends_steer_and_which_fork() -> None: - told = {one.name: one.backends for one in catalogue()} - assert told["steer"] == frozenset(_STEERING) - # Read off `hmz.coganchor.backends` rather than off the drivers: a fork is the CLI's own, so the - # one place a fact about a CLI is written is where the catalogue asks. - assert told["fork"] == frozenset( - name for name in DRIVEN if (one := named(name)) is not None and one.forks - ) - - def test_the_names_a_backend_serves_are_derived_from_its_own_facts() -> None: """`tags` says the vocabulary's word for a fact rather than storing the word too.""" bare = Profile( @@ -273,40 +111,6 @@ def test_each_layer_names_exactly_the_backends_it_reaches() -> None: assert all(bundle.says for bundle in one.bundles), one.name -def test_the_catalogue_names_where_an_agents_turns_may_land() -> None: - told = {one.name: one for one in catalogue()} - for name in ("remote", "isolated", "managed", "linux", "darwin"): - # Every backend, since what a machine is is the same question whichever CLI is - # driven on it -- and what needs saying is said where the place is declared. - assert told[name].backends == frozenset(), name - assert told[name].said - - -def test_an_anchor_nothing_serves_is_left_out_rather_than_read_as_everybodys() -> None: - told = {one.name: one for one in catalogue() if one.name.startswith("anchor:")} - # An empty backend set means every backend here, so a way in that has not been built - # must not be listed at all, and one only some of them serve must be listed with exactly - # those. The three universal ones always are: every backend is a command line spawned - # here, what a spawned turn runs is what an anchor traces, and a line spawned here is a - # line that reads the same on whichever machine it is spawned on. The other two are - # served by the CLIs whose profile says so, and are listed against exactly those. There - # is no `anchor:patched`: nothing takes a turn down that road yet, and a name here would - # be one a flow could ask for and pass. - assert set(told) == { - "anchor:native-cli", - "anchor:supervised", - "anchor:afar", - "anchor:hooked", - "anchor:preloaded", - } - assert told["anchor:native-cli"].backends == frozenset() - assert told["anchor:supervised"].backends == frozenset() - assert told["anchor:hooked"].backends == frozenset({"claude", "qwen"}) - assert told["anchor:preloaded"].backends == frozenset( - {"kimi", "mimo", "pi", "qwen"} - ) - - #: What each backend's driver reports of what a turn cost, written out rather than read off #: the drivers -- what the drivers say is what is on trial. `reasoning` is there only for the #: three that count it beside the output rather than inside it; and two say the input and the @@ -334,114 +138,6 @@ def test_what_each_backend_counts_is_what_its_driver_says_it_counts() -> None: assert cls.counts <= set(KINDS), name -def test_each_kind_of_token_is_a_capability_and_whose_is_the_drivers_own() -> None: - """A flow steering by what a turn cost has to be able to ask before it starts. - - A backend that never counts a kind answers nought for it exactly as one that spent - nothing does, and a loop bounded by an output count on a backend that reports none is a - loop that never ends. - """ - told = {one.name: one for one in catalogue() if one.name.startswith("counts:")} - assert set(told) == {f"counts:{kind}" for kind in KINDS} - for kind in KINDS: - assert told[f"counts:{kind}"].backends == frozenset( - name for name, (cls, _) in DRIVEN.items() if kind in cls.counts - ) - # A kind only some of them count leaves the rest out rather than quietly reading as - # nought for everybody: `reasoning` is the three that count it beside the output. - assert told["counts:reasoning"].backends == frozenset({"agy", "mimo", "opencode"}) - - -def test_every_name_a_backends_own_facts_come_to_is_in_the_catalogue() -> None: - """A name a flow may ask under and a compiler cannot read is one nothing generated asks. - - `search`, `swarm` and `resume` were exactly that. `comes_to` unions a backend's tags in, - so they have always worked in `Needs`; the catalogue and the briefing -- the one page a - compiler steers by -- said nothing about any of them. - """ - named_here = {one.name for one in catalogue()} - - assert {name for one in PROFILES for name in one.tags()} <= named_here - for name in ("search", "swarm", "resume"): - (one,) = (held for held in catalogue() if held.name == name) - tagged = frozenset( - backend - for backend in DRIVEN - if (profile := named(backend)) is not None and name in profile.tags() - ) - - assert one.backends == ( - frozenset() if tagged == frozenset(DRIVEN) else tagged - ), name - # `resume` is every CLI here, so it is named against none of them: an empty set is how - # this catalogue says "all of them", and one listed against twelve names would read as - # something a CLI somebody added by hand does not have. - told = {one.name: one.backends for one in catalogue()} - - assert told["resume"] == frozenset() - assert told["swarm"] == {"kimi"} - - -def test_every_capability_says_which_half_of_needs_asks_for_it() -> None: - """The two halves are answered by different code against different facts. - - A machine capability carries no backends because no backend answers for it, and an empty - backend set is how this catalogue says "every backend here". Without the half written - down, the one is read as the other: `Needs("isolated")` was satisfied by everything. - """ - held = catalogue() - - assert {one.asked for one in held} == {OF_AGENT, WHERE} - asked = {one.name: one.asked for one in held} - for name in ("remote", "isolated", "managed", "linux", "darwin"): - assert asked[name] == WHERE, name - # The roads split between the halves, which is the one thing here that is not obvious. - # An anchor declares the two it reaches a machine by; the two humanize takes from inside - # a process it started are the CLI's own, and no machine has ever carried either. - assert asked["anchor:native-cli"] == WHERE - assert asked["anchor:supervised"] == WHERE - assert asked["anchor:afar"] == WHERE - assert asked["anchor:hooked"] == OF_AGENT - assert asked["anchor:preloaded"] == OF_AGENT - for one in held: - # Backends are meaningless on the machine half and must stay empty there, or the - # emptiness that means "all of them" would read off the wrong side of the line. - if one.asked == WHERE: - assert one.backends == frozenset(), one.name - - -def test_the_places_described_here_are_the_places_coganchor_knows() -> None: - """The words are declared beside the machines and only described by the catalogue.""" - assert {one.name for one in catalogue() if one.asked == WHERE} == ( - places.PLACES | set(places.ROADS) - ) - assert frozenset({"anchor:hooked", "anchor:preloaded"}) == places.INSIDE - - -def test_a_rung_is_named_against_exactly_the_backends_that_take_it() -> None: - """Read off the drivers' own `rungs`, which is what each of them refuses a config for.""" - told = {one.name: one.backends for one in catalogue()} - for permission in PERMISSIONS: - taking = frozenset( - backend for backend, (cls, _) in DRIVEN.items() if permission in cls.rungs - ) - - assert told[rung(permission)] == ( - frozenset() if taking == frozenset(DRIVEN) else taking - ), permission - # dsh is the one driven backend that cannot be held below `bypass`, and it is the only - # one missing from the three narrower rungs. - assert frozenset(DRIVEN) - told[rung("read-only")] == {"dsh"} - # And `bypass` is every backend there is, which the catalogue says by naming none: a CLI - # somebody added by hand is in no `DRIVEN` table, and it takes that rung too. - assert told[rung("bypass")] == frozenset() - - -def test_no_rung_is_named_for_the_silence_above_the_ladder() -> None: - """`UNSAID` is not a rung, and every backend can be told nothing at all.""" - assert rung(UNSAID) not in {one.name for one in catalogue()} - - def test_what_a_driver_says_it_takes_is_what_it_actually_takes() -> None: """`rungs` is a word for a refusal, and a word that drifted from one would be a lie. diff --git a/tests/unit/flows/test_api_surface.py b/tests/unit/flows/test_api_surface.py index c1a30b6f..ac28ca55 100644 --- a/tests/unit/flows/test_api_surface.py +++ b/tests/unit/flows/test_api_surface.py @@ -93,9 +93,16 @@ def _ours(name: str) -> bool: def test_importing_it_asks_for_nothing_of_humanize_and_nothing_heavy( fresh: None, monkeypatch: pytest.MonkeyPatch ) -> None: - """Every import statement the package runs, recorded as it runs, cached or not.""" + """Every import statement the package runs, recorded as it runs, cached or not. + + On this thread alone: a thread another test left running imports what it imports, and + the package importing is what is being watched. + """ + import threading + asked: set[str] = set() importing = builtins.__import__ + here = threading.get_ident() def recorded( name: str, @@ -105,7 +112,9 @@ def recorded( level: int = 0, ) -> Any: package = (globals or {}).get("__package__") - if level and isinstance(package, str): + if threading.get_ident() != here: + pass + elif level and isinstance(package, str): asked.add(importlib.util.resolve_name("." * level + name, package)) else: asked.add(name) diff --git a/tests/unit/flows/test_capabilities.py b/tests/unit/flows/test_capabilities.py deleted file mode 100644 index 4b7b5fe3..00000000 --- a/tests/unit/flows/test_capabilities.py +++ /dev/null @@ -1,442 +0,0 @@ -"""What a flow says filling one of its places takes, and what happens when it does not. - -Most of what a flow builds on, every backend here serves. Some of it only some of them do, -and a flow built on one of those is not a flow any agent can drive. So it writes `Needs` -beside the place, and an agent whose backend serves none of it is refused before the first -turn rather than found out from the call that reached for it, hours into a loop. - -What is covered here is the agent half -- what the backend filling the place has to serve -- -at the top of a run and again where one flow calls another, those being the two places an -agent is ever handed to a flow. The other half, what the machine an agent's turns land on has -to come to, is covered beside the rest of where agents work in `test_where_agents_work.py`. -Nothing here takes a turn: the whole point is that none of it needs one. -""" - -from __future__ import annotations - -from typing import TYPE_CHECKING - -import pytest - -from hmz._legacy_flows import NotAFlow, load -from hmz.coganchor.agents import ( - ClaudeCodeAgent, - ClaudeCodeAgentConfig, - DshAgent, - DshAgentConfig, - Needs, -) -from hmz.runtime.flowing import wanted -from hmz.runtime.runner import Runner -from tests.stubs import written - -if TYPE_CHECKING: - from pathlib import Path - - from hmz.coganchor.backends import Model - -#: A flow that steers the turn it is driving, which only some backends can be asked to do. -STEERS = '''"""One that talks to its agent mid-turn.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, and what filling that place takes.""" - - builder: Annotated[AgentBase, Needs("steer")] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: A flow that asks for three things at once, one of them a moment. -SEVERAL = '''"""One built on more than a flow usually is.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """Two places, only one of which asks for anything.""" - - builder: Annotated[AgentBase, Needs("shape", "tools", "moment:SubagentStop")] - reviewer: AgentBase - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: A flow built on a fact about the CLI rather than on one about the driver that speaks to it. -RESUMES = '''"""One that picks a conversation back up.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives.""" - - builder: Annotated[AgentBase, Needs("resume")] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: One built on something every backend here serves, which is still a thing to be able to say. -EVERYONE = '''"""One built on what nobody has to shop for.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, which has only to do what all of them do.""" - - builder: Annotated[AgentBase, Needs("schema", "hooks")] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: A flow that asks for nothing in particular, which is most flows. -PLAIN = '''"""One that any agent can drive.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - pass -''' - -#: One that calls the one that steers, handing it the agent it was given. -CALLS = '''"""One that reaches for the flow that steers.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow, load - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - load("steers")(agents, task) -''' - -#: One that asks of the *agent* for something only a machine can answer. Legal Python, a real -#: capability name, and the wrong half of `Needs` -- which used to be satisfied by whatever -#: filled the place, a machine capability carrying no backends and no backends meaning all of -#: them. -MISPLACED = '''"""One that asks for a container of the agent rather than of the machine.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, asked for wrongly.""" - - builder: Annotated[AgentBase, Needs("isolated")] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: The same slip the other way round: a road humanize only reaches a turn down from inside a -#: process it started, asked of the machine. No machine's settings have ever carried one, so -#: this was refused by every machine there is -- a check nothing could pass. -INSIDE_OUT = '''"""One that asks a machine for the CLI's own hooks.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, asked for wrongly.""" - - builder: Annotated[AgentBase, Needs(where=("anchor:hooked",))] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: And the same road asked in the half that can answer it, which is the agent's. -HOOKED = '''"""One built on reaching its turns through the CLI's own hooks.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, and what filling that place takes.""" - - builder: Annotated[AgentBase, Needs("anchor:hooked")] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - -#: One whose agent may look and may change nothing, said as a capability rather than only as -#: a setting: a backend with no way of being held to that rung is refused before it is chosen. -READ_ONLY = '''"""One whose reviewer must be holdable to read-only.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Needs -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, and the rung it has to be holdable to.""" - - builder: Annotated[AgentBase, Needs("rung:read-only")] - - -@flow -def run(agents: Agents, task: str) -> None: - pass -''' - - -def _claude() -> ClaudeCodeAgent: - """An agent of the backend whose turns can be talked to while they run.""" - return ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="low")) - - -def _dsh() -> DshAgent: - """An agent of a backend whose turns cannot.""" - return DshAgent(DshAgentConfig(model="m", effort="high")) - - -@pytest.fixture(autouse=True) -def flows(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: - """A project holding the flows these tests drive, and a home nothing wrote to.""" - monkeypatch.setenv("HOME", str(tmp_path / "home")) - where = tmp_path / "project" - kept = where / ".humanize/flows" - kept.mkdir(parents=True) - written(kept, "steers", STEERS) - written(kept, "several", SEVERAL) - written(kept, "resumes", RESUMES) - written(kept, "everyone", EVERYONE) - written(kept, "plain", PLAIN) - written(kept, "calls", CALLS) - written(kept, "misplaced", MISPLACED) - written(kept, "inside_out", INSIDE_OUT) - written(kept, "hooked", HOOKED) - written(kept, "read_only", READ_ONLY) - monkeypatch.chdir(where) - return where - - -def test_a_flow_says_what_filling_each_of_its_places_takes() -> None: - """Read where the agents are chosen, so that only ones that would work are offered.""" - places = wanted("several") - - assert places[0].needs == Needs("shape", "tools", "moment:SubagentStop") - assert places[1].needs is None # which is a place any agent may fill - - -def test_a_place_that_asks_for_nothing_is_a_place_any_backend_may_fill() -> None: - """Most places: what every backend serves is nothing a flow has to write down.""" - Runner("plain", [_dsh()]) # which is the whole assertion: it is not refused - - -def test_a_flow_built_on_steering_is_refused_a_backend_that_cannot_be_steered() -> None: - """Before the first turn, which is the only time refusing it costs nothing.""" - with pytest.raises( - NotAFlow, match="builder has to serve steer, which dsh does not" - ): - Runner("steers", [_dsh()]) - - -def test_a_flow_built_on_steering_takes_a_backend_that_can_be_steered() -> None: - """The other half of the same check: a fit agent is refused nothing.""" - runner = Runner("steers", [_claude()]) - - assert len(runner.agents) == 1 - - -def test_a_flow_is_refused_for_every_one_of_the_things_it_asks_for_at_once() -> None: - """Named together rather than one at a time, so that one reading says what to choose.""" - with pytest.raises(NotAFlow) as refused: - Runner("several", [_dsh(), _dsh()]) - - assert "builder has to serve moment:SubagentStop, shape, tools" in str( - refused.value - ) - assert "which dsh does not" in str(refused.value) - - -def test_what_the_backend_itself_serves_is_read_where_it_is_written_down() -> None: - """`resume` is a fact about the CLI rather than about the driver that speaks to it.""" - runner = Runner("resumes", [_dsh()]) - - assert len(runner.agents) == 1 - - -def test_a_flow_calling_another_is_refused_the_same_way_the_run_would_be() -> None: - """The check is duplicated so that a flow cannot pass at the top and fail in the middle.""" - with pytest.raises( - NotAFlow, match="builder has to serve steer, which dsh does not" - ): - load("calls")([_dsh()], "go") - - -def test_a_flow_calling_another_with_a_fit_agent_is_not_refused() -> None: - """And the called flow runs, which is what being handed a fit agent comes to.""" - load("calls")([_claude()], "go") # the whole assertion: nothing is raised - - -def test_what_every_backend_serves_is_served_by_every_backend() -> None: - """The catalogue names nobody against those, which must not read as nobody serving them.""" - runner = Runner("everyone", [_dsh()]) - - assert len(runner.agents) == 1 - - -def test_a_place_takes_what_it_needs_as_names_rather_than_as_one_name() -> None: - """`where="remote"` is five capabilities spelled a letter each, so it is refused outright.""" - with pytest.raises(TypeError, match="sequence of names rather than one name"): - Needs(where="remote") - - -def test_whoever_is_choosing_an_agent_is_offered_only_the_ones_that_would_do() -> None: - """Asked by backend before there is an agent, so a place cannot be filled wrong.""" - from hmz.runtime.flowing.driving import Place - from hmz.tui.pick import Clis - - offered: dict[str, tuple[Model, ...]] = {"claude": (), "dsh": ()} - plain = Place(name="builder", person=False, moments=frozenset()) - - assert [row[0] for row in Clis(offered, place=plain).rows()] == ["claude", "dsh"] - - steering = plain._replace(needs=Needs("steer")) - - assert [row[0] for row in Clis(offered, place=steering).rows()] == ["claude"] - - -def test_a_place_that_asks_of_the_agent_for_a_machines_answer_is_refused() -> None: - """`Needs("isolated")` used to be satisfied by whatever filled the place. - - A machine capability carries no backends, because no backend answers for it, and an empty - backend set is how the catalogue says "every backend here". So the one slip a type checker - cannot see -- the ask written in the wrong half of `Needs` -- was answered yes by every - agent there is, and a flow that meant to be protected was protected by nothing. It is - refused now, and the refusal says where the ask belongs. - """ - with pytest.raises(NotAFlow, match=r"Needs\(where=\('isolated',\)\)"): - Runner("misplaced", [_claude()]) - - -def test_a_place_that_asks_a_machine_for_the_agents_own_road_is_refused() -> None: - """The same slip the other way, and the one the docstring used to advertise. - - `anchor:hooked` is a hook table humanize writes for one run of a CLI it started here. It - is the CLI's own to take and no machine's settings have ever carried it, so asking for it - under `where=` was refused by every machine there is -- always no, which measures nothing. - """ - with pytest.raises(NotAFlow, match=r"Needs\('anchor:hooked'\)"): - Runner("inside_out", [_claude()]) - - -def test_the_agents_own_road_is_asked_of_the_agent_and_answered_there() -> None: - """Which is the half that can answer: the profile is what declares a hook seam.""" - runner = Runner("hooked", [_claude()]) - - assert len(runner.agents) == 1 - - with pytest.raises(NotAFlow, match="has to serve anchor:hooked"): - Runner("hooked", [_dsh()]) - - -def test_a_flow_may_ask_for_a_rung_before_its_first_turn() -> None: - """The rung a backend can be held to is a capability like any other. - - dsh bundles no confining executor and refuses everything below `bypass`, which it has - always said where the agent is made -- hours after somebody chose it for a flow whose - reviewer may change nothing. Said as a capability, it is said where the choice is made. - """ - runner = Runner("read_only", [_claude()]) - - assert len(runner.agents) == 1 - - with pytest.raises( - NotAFlow, match="has to serve rung:read-only, which dsh does not" - ): - Runner("read_only", [_dsh()]) - - -def test_the_rung_a_backend_refuses_is_the_rung_it_does_not_serve() -> None: - """One fact, read from the driver class and enforced by it, rather than two.""" - from hmz.coganchor.agents import PERMISSIONS, rung - from hmz.runtime.flowing.driving import comes_to - - for permission in PERMISSIONS: - served = rung(permission) in comes_to("dsh") - - assert served is (permission in DshAgent.rungs), permission - if not served: - # And the driver refuses it where the agent is made, which is the fact this - # capability is a word for rather than a second answer beside it. - with pytest.raises(ValueError, match="bypass"): - DshAgent( - DshAgentConfig(model="m", effort="high", permission=permission) - ) - - -def test_the_picker_blames_the_flow_rather_than_the_installation() -> None: - """A place asking of the agent for a machine's answer rules out every CLI there is. - - Which it should -- no backend comes to `isolated` -- but saying that nothing installed - here will do sends somebody off to install a thirteenth CLI for a flow no CLI can fill. - """ - from hmz.runtime.flowing.driving import Place - from hmz.tui.pick import Clis - - offered: dict[str, tuple[Model, ...]] = {"claude": (), "dsh": ()} - wrong = Place( - name="builder", - person=False, - moments=frozenset(), - needs=Needs("isolated"), - ) - picking = Clis(offered, place=wrong) - - assert picking.rows() == [] - assert "Needs(where=('isolated',))" in picking.nothing() - # And a place nothing is wrong with says the other thing, which is still the usual one. - bare = Place(name="builder", person=False, moments=frozenset()) - picking = Clis({}, place=bare) - - assert picking.rows() == [] - assert "no coding agent installed here" in picking.nothing() diff --git a/tests/unit/flows/test_checking.py b/tests/unit/flows/test_checking.py deleted file mode 100644 index 610fa020..00000000 --- a/tests/unit/flows/test_checking.py +++ /dev/null @@ -1,1122 +0,0 @@ -"""The static read of a flow's legality, held to finding what it claims and nothing else. - -Two halves. A fixture or two per rule -- a flow that trips it, and a neighbour standing just -on the legal side -- so that every rule is shown to fire and shown to know where the edge is. -And a sweep over every flow humanize ships and every flow the official flowverse holds: the -rules were written against real flows, and the sweep is the alarm that goes off the day one -of them starts reading a good flow as a bad one. -""" - -from __future__ import annotations - -import textwrap -from typing import TYPE_CHECKING - -import pytest - -import hmz._legacy_flows -from hmz._legacy_flows import Agent, Driven, Person, Session -from hmz.runtime import flowing -from hmz.runtime.flowing import BUILTIN_AT, ENTRY, entry -from hmz.runtime.flowing.checking import checked, offered, surface -from tests.stubs import written - -if TYPE_CHECKING: - from pathlib import Path - -#: What each code is, so that a rule quietly changing its severity is a failing test. -SEVERITY = { - "unread": "error", - "not-a-flow": "error", - "unsized-agents": "error", - "unread-annotation": "error", - "unknown-permission": "error", - "goals-both-ways": "error", - "foreign-import": "error", - "unknown-name": "error", - "unknown-ask": "error", - "dead-loop": "error", - "sleeping-loop": "error", - "stateless-resume": "error", - "unbounded-loop": "warning", - "unguarded-answer": "warning", - "unknown-verdict": "warning", - "unsaid-moment": "warning", - "loose-config": "warning", - "unsaid-field": "warning", - "unsaid-flow": "warning", - "state-kept": "warning", - "twice-named": "warning", -} - -#: What a flow standing just on the legal side of a rule reads as: nothing at all. -CLEAN: set[str] = set() - -#: The one line every fixture flow says about itself, so `unsaid-flow` stays out of the way -#: of every rule but its own. -DOC = '"""A flow written for one rule of the checker."""\n' - -CASES = [ - pytest.param( - DOC - + """ -from pathlib import Path - -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - while True: - Path("beat").write_text(task) -""", - {"dead-loop"}, - id="dead-loop", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - while True: - agent(task, suppress=True) -""", - {"unbounded-loop"}, - id="dead-loop-edge-a-turn-inside-runs-out-of-the-allowance", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - while True: - worked = agent(task, suppress=True) - if not worked: - break -""", - CLEAN, - id="dead-loop-edge-a-break-inside", - ), - pytest.param( - DOC - + """ -import time - -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - while True: - time.sleep(5) -""", - {"sleeping-loop"}, - id="sleeping-loop", - ), - pytest.param( - DOC - + """ -import time - -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - while True: - time.sleep(5) - break -""", - CLEAN, - id="sleeping-loop-edge-it-can-end", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - agent.launch(task) -""", - {"unknown-ask"}, - id="unknown-ask", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - print(agent.spent().output) - session = agent.new() - session(task, suppress=True) - session.close() -""", - CLEAN, - id="unknown-ask-edge-the-interface", - ), - pytest.param( - DOC - + """ -from typing import NamedTuple - -from hmz._legacy_flows import Agent, Person, flow - -class Crew(NamedTuple): - actor: Agent - human: Person - -@flow -def run(agents: Crew, task: str) -> None: - agents.reviewer(task) -""", - {"unknown-ask"}, - id="unknown-ask-a-place-not-declared", - ), - pytest.param( - DOC - + """ -from typing import NamedTuple - -from hmz._legacy_flows import Agent, Person, flow - -class Crew(NamedTuple): - actor: Agent - human: Person - -@flow -def run(agents: Crew, task: str) -> None: - agents.actor(task, suppress=True) - agents.human.board.put("doing", task) -""", - CLEAN, - id="unknown-ask-edge-the-places-declared", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - session = agents[0].new() - session.rewind() -""", - {"unknown-ask"}, - id="unknown-ask-of-a-session", - ), - pytest.param( - DOC - + """ -from hmz.coganchor.agents import Moment - -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0](task) -""", - {"foreign-import"}, - id="foreign-import", - ), - pytest.param( - DOC - + """ -import hmz.coganchor.backends - -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0](task) -""", - {"foreign-import"}, - id="foreign-import-a-module", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, Moment, Usage, backends, flow, home - -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0](task) -""", - CLEAN, - id="foreign-import-edge-the-one-import", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow, teleport - -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0](task) -""", - {"unknown-name"}, - id="unknown-name", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import flow - -@flow -def run(agents, task): - agents[0](task) -""", - {"unsized-agents"}, - id="unsized-agents-nothing-said", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple, task: str) -> None: - agents[0](task) -""", - {"unsized-agents"}, - id="unsized-agents-a-bare-tuple", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent, ...], task: str) -> None: - agents[0](task) -""", - {"unsized-agents"}, - id="unsized-agents-any-number", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent, Agent], task: str) -> None: - agents[0](task) - agents[1](task) -""", - CLEAN, - id="unsized-agents-edge-a-fixed-length", - ), - pytest.param( - DOC - + """ -from typing import Annotated - -from hmz._legacy_flows import Agent, AgentDefaults, flow - -@flow -def run( - agents: tuple[Annotated[Agent, AgentDefaults(permission="readonly")]], task: str -) -> None: - agents[0](task) -""", - {"unknown-permission"}, - id="unknown-permission", - ), - pytest.param( - DOC - + """ -from typing import Annotated - -from hmz._legacy_flows import Agent, AgentDefaults, flow - -@flow -def run( - agents: tuple[Annotated[Agent, AgentDefaults(permission="read-only")]], task: str -) -> None: - agents[0](task) -""", - CLEAN, - id="unknown-permission-edge-a-rung-there-is", - ), - pytest.param( - DOC - + """ -from typing import Annotated - -from hmz._legacy_flows import Agent, AgentDefaults, flow - -@flow -def run( - agents: tuple[Annotated[Agent, AgentDefaults(permission="")]], task: str -) -> None: - agents[0](task) -""", - CLEAN, - id="unknown-permission-edge-nothing-said", - ), - pytest.param( - DOC - + """ -from typing import Annotated - -from hmz._legacy_flows import UNSAID, Agent, AgentDefaults, flow - -@flow -def run( - agents: tuple[Annotated[Agent, AgentDefaults(permission=UNSAID)]], task: str -) -> None: - agents[0](task) -""", - CLEAN, - id="unknown-permission-edge-nothing-said-by-its-name", - ), - pytest.param( - DOC - + """ -from typing import Annotated - -from hmz._legacy_flows import Agent, AgentDefaults, Goal, flow - -@flow -def run( - agents: tuple[Annotated[Agent, Goal, AgentDefaults(goals=False)]], task: str -) -> None: - agents[0].pursue(task) -""", - {"goals-both-ways"}, - id="goals-both-ways", - ), - pytest.param( - DOC - + """ -from typing import Annotated - -from hmz._legacy_flows import Agent, AgentDefaults, Goal, flow - -@flow -def run(agents: tuple[Annotated[Agent, Goal]], task: str) -> None: - agents[0].pursue(task) -""", - CLEAN, - id="goals-both-ways-edge-a-goal-alone", - ), - pytest.param( - DOC - + """ -from typing import TYPE_CHECKING - -from hmz._legacy_flows import flow - -if TYPE_CHECKING: - from hmz._legacy_flows import Agent - -@flow -def run(agents: "tuple[Agent]", task: str) -> None: - agents[0](task) -""", - {"unread-annotation"}, - id="unread-annotation", - ), - pytest.param( - DOC - + """ -from typing import TYPE_CHECKING - -from hmz._legacy_flows import Agent, flow - -if TYPE_CHECKING: - from hmz._legacy_flows import Agent - -@flow -def run(agents: "tuple[Agent]", task: str) -> None: - agents[0](task) -""", - CLEAN, - id="unread-annotation-edge-also-at-runtime", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow(resumable=True) -def run(agents: tuple[Agent], task: str) -> None: - agents[0](task) -""", - {"stateless-resume"}, - id="stateless-resume", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Config(BaseModel): - model_config = {"extra": "forbid"} - - budget: float = Field(default=1.0, description="the bound") - -@flow(resumable=True) -def run(agents: tuple[Agent], task: str, config: Config | None = None) -> None: - agents[0](task) -""", - {"stateless-resume"}, - id="stateless-resume-a-config-in-the-way", - ), - pytest.param( - DOC - + """ -from typing import Any - -from hmz._legacy_flows import Agent, flow - -@flow(resumable=True) -def run(agents: tuple[Agent], task: str, state: dict[str, Any]) -> None: - agents[0](task) -""", - CLEAN, - id="stateless-resume-edge-the-state-third", - ), - pytest.param( - DOC - + """ -import time - -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - done: bool = Field(description="whether it is over") - -@flow -def run(agents: tuple[Agent, Agent], task: str) -> None: - working = agents[0].new() - prompt = task - while True: - worked = working(prompt, suppress=True) - if worked: - review = agents[1](task, suppress=True, schema=Review) - if review is not None and review.done: - return - time.sleep(5) -""", - {"unbounded-loop"}, - id="unbounded-loop", - ), - pytest.param( - DOC - + """ -import time - -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - done: bool = Field(description="whether it is over") - -@flow -def run(agents: tuple[Agent, Agent], task: str) -> None: - working = agents[0].new() - while True: - working(task, suppress=True) - review = agents[1](task, suppress=True, schema=Review) - if review is not None and review.done: - return - if agents[0].spent().output > 1_000_000: - return - time.sleep(5) -""", - CLEAN, - id="unbounded-loop-edge-a-budget-bound", - ), - pytest.param( - DOC - + """ -import time - -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - (agent,) = agents - while True: - said = agent(task, suppress=True) - if said: - return - time.sleep(5) -""", - CLEAN, - id="unbounded-loop-edge-a-turn-merely-landing", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - done: bool = Field(description="whether it is over") - -@flow -def run(agents: tuple[Agent], task: str) -> None: - review = agents[0](task, suppress=True, schema=Review) - print(review.done) -""", - {"unguarded-answer"}, - id="unguarded-answer", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - done: bool = Field(description="whether it is over") - -@flow -def run(agents: tuple[Agent], task: str) -> None: - review = agents[0](task, suppress=True, schema=Review) - if review is not None and review.done: - print("done") -""", - CLEAN, - id="unguarded-answer-edge-guarded", - ), - pytest.param( - DOC - + """ -from typing import Literal - -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - verdict: Literal["done", "redo"] = Field(description="how the round went") - -@flow -def run(agents: tuple[Agent], task: str) -> None: - review = agents[0](task, suppress=True, schema=Review) - if review and review.verdict == "DONE": - return -""", - {"unknown-verdict"}, - id="unknown-verdict", - ), - pytest.param( - DOC - + """ -from typing import Literal - -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - verdict: Literal["done", "redo"] = Field(description="how the round went") - -@flow -def run(agents: tuple[Agent], task: str) -> None: - review = agents[0](task, suppress=True, schema=Review) - if review and review.verdict in ("done", "OOPS"): - return -""", - {"unknown-verdict"}, - id="unknown-verdict-a-dead-member", - ), - pytest.param( - DOC - + """ -from typing import Literal - -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - verdict: Literal["done", "redo"] = Field(description="how the round went") - -@flow -def run(agents: tuple[Agent], task: str) -> None: - review = agents[0](task, suppress=True, schema=Review) - if review and review.verdict == "done": - return -""", - CLEAN, - id="unknown-verdict-edge-a-value-the-shape-offers", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Review(BaseModel): - model_config = {"extra": "forbid"} - - verdict: str = Field(description="how the round went, in its own words") - -@flow -def run(agents: tuple[Agent], task: str) -> None: - review = agents[0](task, suppress=True, schema=Review) - if review and review.verdict == "DONE": - return -""", - CLEAN, - id="unknown-verdict-edge-a-field-left-open", - ), - pytest.param( - DOC - + """ -from _shapes import Review - -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - review = agents[0](task, suppress=True, schema=Review) - if review and review.verdict == "DONE": - return -""", - CLEAN, - id="unknown-verdict-edge-a-shape-declared-elsewhere", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, Moment, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0].hooks.on(Moment.PERMISSION_REQUEST, print) -""", - {"unsaid-moment"}, - id="unsaid-moment", - ), - pytest.param( - DOC - + """ -from typing import Annotated, NamedTuple - -from hmz._legacy_flows import Agent, Moment, flow - -class Crew(NamedTuple): - builder: Annotated[Agent, Moment.PERMISSION_REQUEST] - -@flow -def run(agents: Crew, task: str) -> None: - agents.builder.hooks.on(Moment.PERMISSION_REQUEST, print) -""", - CLEAN, - id="unsaid-moment-edge-the-place-declares-it", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, Moment, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0].hooks.on(Moment.STOP, print) -""", - CLEAN, - id="unsaid-moment-edge-a-moment-every-backend-runs", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Config(BaseModel): - budget: float = Field(default=1.0, description="the bound") - -@flow -def run(agents: tuple[Agent], task: str, config: Config | None = None) -> None: - agents[0](task) -""", - {"loose-config"}, - id="loose-config", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel, Field - -class Config(BaseModel): - model_config = {"frozen": True} - - budget: float = Field(default=1.0, description="the bound") - -@flow -def run(agents: tuple[Agent], task: str, config: Config | None = None) -> None: - agents[0](task) -""", - CLEAN, - id="loose-config-edge-frozen", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel - -class Config(BaseModel): - model_config = {"extra": "forbid"} - - budget: float - -@flow -def run(agents: tuple[Agent], task: str, config: Config | None = None) -> None: - agents[0](task) -""", - {"unsaid-field"}, - id="unsaid-field", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow -from pydantic import BaseModel - -class Answer(BaseModel): - said: str - -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0](task, suppress=True, schema=Answer) -""", - CLEAN, - id="unsaid-field-edge-a-schema-is-not-a-config", - ), - pytest.param( - """ -from hmz._legacy_flows import Agent, flow - -@flow -def run(agents: tuple[Agent], task: str) -> None: - agents[0](task) -""", - {"unsaid-flow"}, - id="unsaid-flow", - ), - pytest.param( - DOC - + """ -from typing import Any - -from hmz._legacy_flows import Agent, flow - -@flow(resumable=True) -def run(agents: tuple[Agent], task: str, state: dict[str, Any]) -> None: - kept = state - kept["rounds"] = kept.get("rounds", 0) + 1 - agents[0](task) -""", - {"state-kept"}, - id="state-kept", - ), - pytest.param( - DOC - + """ -from typing import Any - -from hmz._legacy_flows import Agent, flow - -@flow(resumable=True) -def run(agents: tuple[Agent], task: str, state: dict[str, Any]) -> None: - kept = state - kept["rounds"] = kept.get("rounds", 0) + 1 - agents[0](task) - kept.clear() -""", - CLEAN, - id="state-kept-edge-cleared", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow(name="draft") -def one(agents: tuple[Agent], task: str) -> None: - agents[0](task) - -@flow(name="draft") -def two(agents: tuple[Agent], task: str) -> None: - agents[0](task) -""", - {"twice-named"}, - id="twice-named", - ), - pytest.param( - DOC - + """ -from hmz._legacy_flows import Agent, flow - -@flow(name="draft") -def one(agents: tuple[Agent], task: str) -> None: - agents[0](task) - -@flow(name="check") -def two(agents: tuple[Agent], task: str) -> None: - agents[0](task) -""", - CLEAN, - id="twice-named-edge-two-names", - ), - pytest.param( - DOC - + """ -def run(agents, task): - return None -""", - {"not-a-flow"}, - id="not-a-flow", - ), - pytest.param( - "def run(:\n", - {"unread"}, - id="unread", - ), -] - - -@pytest.mark.parametrize(("source", "expected"), CASES) -def test_each_rule_fires_and_knows_the_edge( - tmp_path: Path, source: str, expected: set[str] -) -> None: - at = written(tmp_path, "one", textwrap.dedent(source)) - found = checked(at) - assert {one.code for one in found} == expected, found - for one in found: - assert one.severity == SEVERITY[one.code] - assert one.line >= 0 - - -def test_a_flow_that_is_not_there_is_not_a_flow(tmp_path: Path) -> None: - found = checked(tmp_path / "nowhere") - assert [one.code for one in found] == ["not-a-flow"] - - -def test_a_single_file_flow_is_read_as_one(tmp_path: Path) -> None: - at = tmp_path / "alone.py" - at.write_text( - textwrap.dedent( - ''' - """A flow that is one file.""" - - from hmz._legacy_flows import Agent, flow - - @flow - def run(agents: tuple[Agent], task: str) -> None: - while True: - agents[0](task, suppress=True) - ''' - ) - ) - assert [one.code for one in checked(at)] == ["unbounded-loop"] - - -def test_what_is_under_skills_is_not_read(tmp_path: Path) -> None: - """A skill may ship a helper script, and it is the agents' content, not the flow's.""" - at = written( - tmp_path, - "one", - DOC - + textwrap.dedent( - """ - from hmz._legacy_flows import Agent, flow - - @flow - def run(agents: tuple[Agent], task: str) -> None: - agents[0](task) - """ - ), - skills={"helping": "# How to help\n"}, - ) - beside = at / "skills" / "helping" / "helper.py" - beside.write_text("import hmz.coganchor.backends\nwhile True:\n pass\n") - assert checked(at) == () - - -def test_a_finding_says_which_file_beside_the_entry_it_is_in(tmp_path: Path) -> None: - at = written( - tmp_path, - "one", - DOC - + textwrap.dedent( - """ - from hmz._legacy_flows import Agent, flow - - @flow - def run(agents: tuple[Agent], task: str) -> None: - agents[0](task) - """ - ), - ) - (at / "helper.py").write_text( - textwrap.dedent( - ''' - """What the flow imports beside itself.""" - - def churn() -> None: - while True: - print("round and round") - ''' - ) - ) - found = checked(at) - assert [(one.code, one.where.name) for one in found] == [("dead-loop", "helper.py")] - - -def test_the_surface_is_the_interfaces_themselves() -> None: - """What an agent may be asked is read off `agent.py`, not kept as a second list.""" - assert {"new", "clone", "spent", "hooks", "epic", "__call__"} <= surface(Agent) - assert "board" not in surface(Agent) - assert "board" in surface(Person) - assert {"loads", "close", "stream"} <= surface(Session) - assert "rename" not in surface(Agent) - assert "rename" in surface(Driven) - - -def test_nothing_said_is_a_permission_a_flow_may_write() -> None: - """The word for no rung at all is one a flow reaches for, so the facade hands it over.""" - from hmz.coganchor.agents import UNSAID - - assert hmz._legacy_flows.UNSAID is UNSAID - assert "UNSAID" in hmz._legacy_flows.__all__ - assert UNSAID not in hmz._legacy_flows.PERMISSIONS - - -def test_a_rung_there_is_not_says_what_there_is(tmp_path: Path) -> None: - """The finding names the ladder and the silence above it, both being writable.""" - at = written( - tmp_path, - "one", - textwrap.dedent( - DOC - + """ -from typing import Annotated - -from hmz._legacy_flows import Agent, AgentDefaults, flow - -@flow -def run( - agents: tuple[Annotated[Agent, AgentDefaults(permission="rdonly")]], task: str -) -> None: - agents[0](task) -""" - ), - ) - (said,) = [one for one in checked(at) if one.code == "unknown-permission"] - assert "'rdonly' is no rung there is" in said.said - for rung in hmz._legacy_flows.PERMISSIONS: - assert rung in said.said - assert "said nothing" in said.said - - -def test_everything_offered_is_reachable() -> None: - """`offered` is what `unknown-name` trusts, so a name in it nothing answers is a lie.""" - said = offered() - assert {"flow", "Agent", "Moment", "home", "models", "backends"} <= said - assert "ClaudeCodeAgent" not in said - for name in sorted(said): - assert getattr(hmz._legacy_flows, name, None) is not None, name - - -#: Every warning a flow humanize ships or the official flowverse holds is allowed to keep. -#: rlar's loop is ended by its reviewer alone, which is the flow's own documented shape -- -#: and exactly the shape the checker exists to point at, so the warning stands. -ALLOWED_WARNINGS = { - "rlar": {"unbounded-loop"}, - # Four loops whose only end is the run's allowance being spent. That is legal -- a turn - # taken once it is spent raises -- and it is still worth saying, because such a loop stops - # rather than finishes and how long it takes is a number somebody else set. - "continue_loop": {"unbounded-loop"}, - "fixed_juice_ralph": {"unbounded-loop"}, - "flame_chase": {"unbounded-loop"}, - "goal": {"unbounded-loop"}, - # And two that do decide to stop -- three rounds answering with nothing -- and keep what - # they kept when they do, on purpose: a loop that stalled is one to fix and carry on from. - # Which is the judgement `state-kept` exists to make somebody state, and they state it. - "ralph_loop": {"state-kept"}, - "stateful_ralph": {"state-kept"}, -} - - -def _swept() -> list[object]: - """One parameter per flow humanize ships or the official flowverse holds now.""" - places = [("package", BUILTIN_AT)] - places.extend( - ("official", under) - for verse in flowing.flowverses() - if verse.name == flowing.OFFICIAL and verse.fetched - for under in flowing.holds(verse) - if under != BUILTIN_AT - ) - held: list[object] = [] - seen_official = False - for whose, under in places: - seen_official = seen_official or whose == "official" - for name in flowing.offered(under): - at = entry(under, name) - if at is None: - continue - target = at.parent if at.name == ENTRY else at - # What the official flowverse holds imports `hmz.flows`, which names the new flow - # API now; this checker reads the old one, kept at `hmz._legacy_flows` until the - # flowverse has moved over and the checker is deleted with it. - marks = ( - [ - pytest.mark.skip( - reason=f"{name} imports hmz.flows, the new flow API now" - ) - ] - if whose == "official" - else [] - ) - held.append(pytest.param(target, name, id=f"{whose}/{name}", marks=marks)) - if not seen_official: - held.append( - pytest.param( - None, - "official", - id="official/unfetched", - marks=pytest.mark.skip( - reason="the official flowverse has not been fetched here" - ), - ) - ) - return held - - -@pytest.mark.parametrize(("at", "name"), _swept()) -def test_every_flow_humanize_offers_reads_clean(at: Path, name: str) -> None: - """The false-positive alarm: real flows, and exactly the warnings they are allowed.""" - found = checked(at) - errors = [one for one in found if one.severity == "error"] - assert not errors, errors - warned = {one.code for one in found if one.severity == "warning"} - assert warned == ALLOWED_WARNINGS.get(name, set()), found diff --git a/tests/unit/flows/test_engine_calls.py b/tests/unit/flows/test_engine_calls.py index 1c309384..d2af3f35 100644 --- a/tests/unit/flows/test_engine_calls.py +++ b/tests/unit/flows/test_engine_calls.py @@ -463,6 +463,57 @@ async def outer( assert said == ["/home/me/project", "/elsewhere"] +@pytest.mark.parametrize("declared", [HereEnvs, HereShellEnvs]) +async def test_a_local_role_refuses_an_environment_on_another_machine( + declared: type[EnvCollection], +) -> None: + """A `LocalEnv` is this machine: one on an ssh host is refused, down either path.""" + + @flow(agents=AgentCollection, envs=declared, params=Nothing) + async def inner( + task: str, + *, + agents: AgentCollection, + envs: EnvCollection, + params: Nothing, + ctx: FlowContext, + ) -> str: + return str(envs["here"].workdir) + + class Shell(Env, ShellEnvMixin): ... + + class Two(EnvCollection): + remote: Shell + local: Shell + + @flow(agents=AgentCollection, envs=Two, params=Nothing) + async def outer( + task: str, + *, + agents: AgentCollection, + envs: Two, + params: Nothing, + ctx: FlowContext, + ) -> str: + for _ in range(2): # the second time a grant is met is the fast path's + with pytest.raises(CapabilityMissing, match="not this machine"): + await inner( + task, agents={}, envs={"here": envs["remote"]}, params=Nothing() + ) + return await inner( + task, agents={}, envs={"here": envs["local"]}, params=Nothing() + ) + + said = await run_fake( + outer, + envs={ + "remote": FakeEnvDriver(workdir="/far", backend="ssh", provider="box"), + "local": FakeEnvDriver(workdir="/near"), + }, + ) + assert said == "/near" + + async def test_a_local_role_cannot_be_given_a_driver() -> None: @flow(agents=AgentCollection, envs=HereEnvs, params=Nothing) async def here( diff --git a/tests/unit/flows/test_flow_interface.py b/tests/unit/flows/test_flow_interface.py deleted file mode 100644 index 55584224..00000000 --- a/tests/unit/flows/test_flow_interface.py +++ /dev/null @@ -1,140 +0,0 @@ -"""The one import a flow writes, and what answers to it. - -Three things nothing else checks. That `hmz._legacy_flows` really is the whole of what a flow -needs -- -which is only true while the flows humanize itself ships name nothing else, since they are the -worked example every other flow is copied from. That everything it says it offers is reachable, -the vocabulary being handed through by name rather than imported. And that the drivers answer -to the interfaces a flow is written against, which is stated for a type checker where the -interfaces are declared and is worth having said once where a run of the suite can hear it. -""" - -from __future__ import annotations - -import ast -from typing import TYPE_CHECKING - -import pytest - -import hmz._legacy_flows -import hmz.coganchor.agents -from hmz._legacy_flows import Agent, Person, Session -from hmz._legacy_flows import Unrecoverable as FlowUnrecoverable -from hmz.coganchor.agents import HumanAgent -from hmz.coganchor.agents import Unrecoverable as AgentUnrecoverable -from hmz.runtime.flowing import BUILTIN_AT -from hmz.runtime.flowing.checking import surface - -if TYPE_CHECKING: - from collections.abc import Iterator - from pathlib import Path - -#: What a flow may name of humanize's own, which is one thing. Everything else it needs -- -#: the vocabulary a turn is described in, the facts about the CLIs, where humanize keeps what -#: outlives a run -- is handed through from there. -ONLY = "hmz._legacy_flows" - -#: What a flow names for the two things that reach back out of a turn: the callbacks it puts -#: in front of an agent as tools, and the board it and the person both write on. Each is -#: written in `hmz.coganchor.agents`. `Tool` is in an import line in the guide that introduces it; -#: `Board`, `Item` and `Refused` are the types and the exception a flow meets through -#: `person.board`, which it has to be able to annotate and catch by name. -REACHING = ("Board", "Item", "Refused", "Tool") - - -def _flows() -> Iterator[Path]: - """Every Python file in the flows humanize itself ships.""" - return BUILTIN_AT.rglob("*.py") - - -def _named(source: Path) -> set[str]: - """Every module of humanize's own that one file names in an import.""" - said: set[str] = set() - for node in ast.walk(ast.parse(source.read_text(encoding="utf-8"))): - if isinstance(node, ast.Import): - said.update(alias.name for alias in node.names) - elif isinstance(node, ast.ImportFrom) and node.module: - said.add(node.module) - return {one for one in said if one.split(".")[0] == "hmz"} - - -@pytest.mark.parametrize("source", sorted(_flows()), ids=lambda one: one.parent.name) -def test_a_flow_humanize_ships_names_nothing_of_humanize_but_hmz_flows( - source: Path, -) -> None: - """These are the worked example; a flow copied from one inherits what it imports.""" - assert _named(source) <= {ONLY} - - -@pytest.mark.parametrize("name", sorted(hmz._legacy_flows.__all__)) -def test_everything_it_offers_is_there(name: str) -> None: - """A name in `__all__` with nothing behind it is an import that fails at the first run.""" - assert getattr(hmz._legacy_flows, name, None) is not None - - -def test_a_name_it_does_not_offer_is_an_attribute_error() -> None: - """Handing names through must not turn a typo into something that is silently None.""" - with pytest.raises(AttributeError): - _ = hmz._legacy_flows.ClaudeCodeAgent # type: ignore[attr-defined] - - -def test_an_unrecoverable_turn_is_the_same_exception_a_flow_can_catch() -> None: - """The flow-facing vocabulary is handed through, not redefined at the boundary.""" - assert FlowUnrecoverable is AgentUnrecoverable - - -@pytest.mark.parametrize("name", REACHING) -def test_what_reaches_back_out_of_a_turn_is_offered_by_name(name: str) -> None: - """A flow names each of these, and offered means both things. - - In `__all__`, since that is where a type checker reads what this module lets through -- - a name only `__getattr__` knows about is a name a flow imports and then cannot call -- - and the same object, since the tool the flow builds is the tool a backend is handed. - """ - assert name in hmz._legacy_flows.__all__ - assert getattr(hmz._legacy_flows, name) is getattr(hmz.coganchor.agents, name) - - -def test_everything_handed_through_is_offered() -> None: - """A name `__getattr__` knows and `__all__` does not is `object` to a type checker. - - Which is the direction nothing else watches: pyright reads a name in `__all__` that is - nowhere in the module and says so, and says nothing at all about one handed through and - never listed. So the list above is these four names being right today, and this is the - rule they are right by. - """ - handed = set(hmz._legacy_flows._ELSEWHERE) | set(hmz._legacy_flows._MODULES) - assert handed <= set(hmz._legacy_flows.__all__) - - -def _answers(driver: object) -> set[str]: - """What one driver has, by name. - - Args: - driver: A driver, made rather than named: an attribute written in `__init__` -- the - epic the run is written into is one -- is on the agent rather than on its class, and - is as much what the driver has as a method is. - - Returns: - Everything reachable on it. - """ - return set(dir(driver)) - - -def test_the_drivers_answer_to_what_a_flow_drives() -> None: - """Structurally, since the arrow points one way: `hmz.coganchor.agents` never names a flow. - - Against a driver that was made rather than against the classes, and against the one driver - that can be made without a coding agent behind it: the person is an `AgentBase` and their - turn is a `SessionBase`, so what is checked here is the pair every backend is driven as. - """ - person = HumanAgent() - for interface, driver in ( - (Agent, person), - (Person, person), - (Session, person.new()), - ): - missing = {one for one in surface(interface) if one not in _answers(driver)} - assert not missing, ( - f"{type(driver).__name__} does not answer to {interface.__name__}: {missing}" - ) diff --git a/tests/unit/flows/test_goals.py b/tests/unit/flows/test_goals.py deleted file mode 100644 index a0b2a5a8..00000000 --- a/tests/unit/flows/test_goals.py +++ /dev/null @@ -1,227 +0,0 @@ -"""A flow built on the backends' own goal feature says so, and is refused an agent without one. - -`pursue` is the agent keeping itself going toward an objective it decides for itself is met, -and four of the seven backends have it. A flow written around that is not a flow any agent can -drive -- so it declares it where it declares the agent, exactly as it declares a moment it -hangs a hook on, and a run that could not work is refused before its first turn rather than -raising in the middle of one. -""" - -from __future__ import annotations - -from typing import TYPE_CHECKING - -import pytest - -from hmz._legacy_flows import NotAFlow -from hmz.coganchor.agents import ( - ClaudeCodeAgent, - ClaudeCodeAgentConfig, - CodexAgent, - CodexAgentConfig, - DshAgent, - DshAgentConfig, - KimiCodeCLIAgent, - KimiCodeCLIAgentConfig, - OpencodeAgent, - OpencodeAgentConfig, - PiAgent, - PiAgentConfig, -) -from hmz.runtime.flowing import wanted -from hmz.runtime.runner import Runner, flow_and_agents - -if TYPE_CHECKING: - from pathlib import Path - -#: A flow that runs its one agent under a goal, and says so where it declares it. -PURSUING = '''"""A loop that hands the objective to the agent and lets it decide it is done.""" - -from typing import Annotated, NamedTuple - -from hmz.coganchor.agents import AgentBase, Goal -from hmz._legacy_flows import flow - - -class Agents(NamedTuple): - """The one it drives, which has to have a goal feature of its own.""" - - worker: Annotated[AgentBase, Goal] - - -@flow -def run(agents: Agents, task: str) -> None: - agents.worker.pursue(task) -''' - -#: The same loop, said the ordinary way: turns, and nothing asked of the backend. -PLAIN = '''"""A loop of plain turns, which any backend takes.""" - -from hmz.coganchor.agents import AgentBase -from hmz._legacy_flows import flow - - -@flow -def run(agents: tuple[AgentBase], task: str) -> None: - (agent,) = agents - agent(task) -''' - -#: The same ordinary loop, declared without goals: this one owns its continuations, and -#: nobody outside it has any say in that. -GOALS_OFF = '''"""A loop that owns its continuations, and says so where it declares its agent.""" - -from typing import Annotated - -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow - - -@flow -def run( - agents: tuple[Annotated[AgentBase, AgentDefaults(goals=False)]], task: str -) -> None: - (agent,) = agents - agent(task) -''' - -#: A required goal is the more particular of the two things a flow can write, and wins. -#: Written both ways at once is a flow the checker has something to say about; read back, it -#: is a place run under a goal. -REQUIRED_WHILE_OFF = PURSUING.replace( - "from hmz.coganchor.agents import AgentBase, Goal", - "from hmz.coganchor.agents import AgentBase, AgentDefaults, Goal", -).replace( - "Annotated[AgentBase, Goal]", - "Annotated[AgentBase, Goal, AgentDefaults(goals=False)]", -) - - -def _written(tmp_path: Path, source: str, name: str = "pursuing") -> str: - """Writes a flow out and answers with its path.""" - where = tmp_path / f"{name}.py" - where.write_text(source) - return str(where) - - -def test_a_place_run_under_a_goal_says_so(tmp_path: Path) -> None: - """Which is what whoever is choosing the agents reads, before anything runs.""" - (place,) = wanted(_written(tmp_path, PURSUING)) - - assert place.goal is True - assert place.name == "worker" - - -def test_a_place_that_said_nothing_is_driven_by_turns(tmp_path: Path) -> None: - (place,) = wanted(_written(tmp_path, PLAIN, "plain")) - - assert place.goal is False - assert place.goals is True - - -def test_a_place_declares_whether_its_agent_has_goals(tmp_path: Path) -> None: - """Which is the flow's to say: nobody else is asked, so nobody else can answer.""" - (place,) = wanted(_written(tmp_path, GOALS_OFF, "goals_off")) - - assert place.goal is False - assert place.goals is False - - -def test_a_required_goal_is_a_goal(tmp_path: Path) -> None: - (place,) = wanted(_written(tmp_path, REQUIRED_WHILE_OFF, "required")) - - assert place.goal is True - assert place.goals is True - - -def test_an_agent_whose_backend_has_no_goal_feature_is_refused(tmp_path: Path) -> None: - """Before the first turn, which is where a loop would otherwise find out.""" - where = _written(tmp_path, PURSUING) - - with pytest.raises( - NotAFlow, match="is run under a goal, which pi has no feature for" - ): - Runner(where, [PiAgent(PiAgentConfig(model="m", effort="low"))]) - - with pytest.raises(NotAFlow, match="opencode has no feature for"): - Runner(where, [OpencodeAgent(OpencodeAgentConfig(model="m", effort="low"))]) - - -@pytest.mark.parametrize( - "agent", - [ - ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="low")), - CodexAgent(CodexAgentConfig(model="m", effort="low")), - DshAgent(DshAgentConfig(model="m", effort="high")), - KimiCodeCLIAgent(KimiCodeCLIAgentConfig(model="m", effort="low")), - ], -) -def test_an_agent_whose_backend_has_one_is_taken(agent: object, tmp_path: Path) -> None: - """The four that answer `pursue`, each of which says so on the class.""" - runner = Runner(_written(tmp_path, PURSUING), [agent]) # pyright: ignore[reportArgumentType] - - assert len(runner.agents) == 1 - - -def test_a_flow_that_asks_nothing_takes_any_of_them(tmp_path: Path) -> None: - """A place that says nothing about a goal is one every backend can fill.""" - runner = Runner( - _written(tmp_path, PLAIN, "plain"), - [PiAgent(PiAgentConfig(model="m", effort="low"))], - ) - - assert len(runner.agents) == 1 - - -def test_the_runner_settles_goals_from_what_the_flow_declared(tmp_path: Path) -> None: - """Over whatever the agent was made with: the flow says, and it says for the run.""" - agent = ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="low")) - - Runner(_written(tmp_path, GOALS_OFF, "goals_off"), [agent]) - - assert agent.config.goals is False - assert not agent.goals_enabled - - -def test_a_flow_that_says_nothing_leaves_its_agent_as_it_came(tmp_path: Path) -> None: - """Goals on is the loosest of the two, and a declaration only ever tightens. - - Which is what keeps a call from handing itself more than it was given: a flow that - mentions nothing declares the loosest of each, and if that settled anything then calling - one would switch goals back on for an agent somebody switched them off for. - """ - agent = ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="low", goals=False)) - - Runner(_written(tmp_path, PLAIN, "plain"), [agent]) - - assert agent.config.goals is False - assert not agent.goals_enabled - - -def test_an_exec_line_leaves_goals_to_the_flow(tmp_path: Path) -> None: - """The line names a CLI, a model and an effort; the flow says what the work is.""" - where = _written(tmp_path, GOALS_OFF, "goals_off") - - _, agents, _, _, _, _ = flow_and_agents( - ["-f", where, "-a", "claude/m:low", "the task"] - ) - assert agents[0].config.goals is True - - Runner(where, agents) - assert agents[0].config.goals is False - - -def test_a_required_goal_cannot_be_switched_off(tmp_path: Path) -> None: - agent = ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="low", goals=False)) - - with pytest.raises(NotAFlow, match="run under a goal, but goals were switched off"): - Runner(_written(tmp_path, PURSUING), [agent]) - - -def test_a_required_goal_refuses_an_agent_disabled_in_python(tmp_path: Path) -> None: - agent = ClaudeCodeAgent(ClaudeCodeAgentConfig(model="m", effort="low")) - agent.disable_goals() - - assert agent.config.goals is False - with pytest.raises(NotAFlow, match="run under a goal, but goals were switched off"): - Runner(_written(tmp_path, PURSUING), [agent]) diff --git a/tests/unit/flows/test_leniency.py b/tests/unit/flows/test_leniency.py deleted file mode 100644 index c535c1d4..00000000 --- a/tests/unit/flows/test_leniency.py +++ /dev/null @@ -1,374 +0,0 @@ -"""What a place gets when it would rather run than be refused, and what it is still refused. - -A declaration is meant to hold, so a backend that cannot carry one refuses the run -- that is -the doctrine, it is the default, and it stays the default. `insist=False` is the other honest -answer to one particular problem: a benchmark declaring `web_search=False` across every CLI -there is is declaring something about the comparison rather than about the roster, and the -three backends with no way of being told should cost it three noisier cells rather than three -cells that never ran. - -What it must never be is "apply it anyway and say nothing". The setting that cannot be carried -is dropped -- the config comes out not carrying it -- the rest are settled, and a line says -which place, which setting and what the agent does instead. A config left holding -`web_search=False` on an agent that will go on searching is the whole of what this seam exists -to prevent, and asking politely does not make it true. - -Here rather than through a flow because none of it needs one: a `Place` is a named tuple, an -agent is an object, and `runs_at` is the one call between them. Nothing is started, nothing is -installed and no CLI is run. -""" - -from __future__ import annotations - -from typing import TYPE_CHECKING, Any - -import pytest - -from hmz.coganchor.agents import ( - UNSAID, - AgentBase, - AgentConfig, - AgentDefaults, - AntigravityCLIAgent, - ClaudeCodeAgent, - PiAgent, - Unserved, -) -from hmz.runtime.flowing.driving import NotAFlow, Place, _declared, runs_at -from hmz.runtime.runner import Runner - -if TYPE_CHECKING: - from pathlib import Path - -#: The backends whose profiles say they cannot be told about the web at all, which is what a -#: flow declaring `web_search=False` across a roster runs into. `pi` is the soundest of them -#: and the one the rest of this file leans on: it has tool control its driver already uses and -#: simply no web tool to withhold, so there is nothing there for anybody to switch off later. -#: `agy` is the other. `dsh` and `kimi` were on this list until their drivers learned to say -#: it -- which is the right way to answer a backend that cannot be told, and is why leniency -#: is the second answer rather than the first. -UNTELLABLE = (PiAgent, AntigravityCLIAgent) - - -def place(**said: Any) -> Place: - """A place called `reviewer` that declares what it is given and nothing else.""" - return Place(name="reviewer", person=False, moments=frozenset(), **said) - - -def agent(kind: type[AgentBase], **said: Any) -> AgentBase: - """One of those agents, made with a model and an effort nothing here reads.""" - return kind(AgentConfig(model="a-model", effort="", **said)) - - -class TestTheFlowbenchCase: - """A place declaring `web_search=False` onto a backend with no way of being told.""" - - @pytest.mark.parametrize("kind", UNTELLABLE) - def test_it_runs_rather_than_being_refused(self, kind: type[AgentBase]) -> None: - """Which is the point: the cell runs, where before it never started.""" - runs_at("bench.py", agent(kind), place(web_search=False, insist=False)) - - @pytest.mark.parametrize("kind", UNTELLABLE) - def test_the_config_does_not_come_out_carrying_the_answer_it_cannot_keep( - self, kind: type[AgentBase] - ) -> None: - """The honest half. An agent that will search must not be holding `False`. - - This is the assertion the whole design is for. Leniency that left the declaration on - the config would be leniency that made the config lie, which is worse than the - refusal it replaced -- the refusal at least said so. - """ - one = agent(kind) - runs_at("bench.py", one, place(web_search=False, insist=False)) - assert one.config.web_search is not False - - @pytest.mark.parametrize("kind", UNTELLABLE) - def test_it_says_which_place_which_setting_and_what_happens_instead( - self, kind: type[AgentBase] - ) -> None: - """The other honest half: dropped is not the same as ignored unless nobody is told.""" - dropped: list[str] = [] - runs_at( # type: ignore[arg-type] - "bench.py", - agent(kind), - place(web_search=False, insist=False), - dropped=dropped, - ) - assert len(dropped) == 1 - assert "reviewer" in dropped[0] - assert "web_search" in dropped[0] - assert "reading the web" in dropped[0] - - @pytest.mark.parametrize("kind", UNTELLABLE) - def test_insisting_is_still_the_refusal_it_always_was( - self, kind: type[AgentBase] - ) -> None: - """A place that says nothing about insisting insists, so nothing moved under anyone.""" - with pytest.raises( - NotAFlow, match="no way of being told not to search the web" - ): - runs_at("bench.py", agent(kind), place(web_search=False)) - - -class TestTheRestIsStillSettled: - """Dropping one declaration is not dropping the declaration.""" - - def test_the_servable_one_settles_and_the_other_is_dropped(self) -> None: - """Two declared, one carried: pi takes a rung and cannot be told about the web. - - `read-only` it serves, out of the `--exclude-tools` its driver already writes; the - web it has no tool for at all, so there is nothing to withhold and nothing to say. - One declaration settles, the other is given up, and the line names only the one that - was given up. - """ - one = agent(PiAgent) - dropped: list[str] = [] - runs_at( - "bench.py", - one, - place(web_search=False, permission="read-only", insist=False), - dropped=dropped, - ) - assert one.config.permission == "read-only" - assert one.config.web_search is not False - assert "web_search" in dropped[0] - assert "permission" not in dropped[0] - - def test_a_backend_that_carries_both_drops_neither(self) -> None: - """Leniency is reached for only where something was actually refused.""" - one = agent(ClaudeCodeAgent) - dropped: list[str] = [] - runs_at( - "bench.py", - one, - place(web_search=False, permission="read-only", insist=False), - dropped=dropped, - ) - assert one.config.web_search is False - assert one.config.permission == "read-only" - assert dropped == [] - - def test_two_refusals_take_two_passes_and_both_are_given_up(self) -> None: - """`_serves` raises at the first thing it finds, so settling asks again. - - agy refuses the web because its profile says it cannot be told, and refuses - `read-only` beside `disable_slash_commands` because its plan mode does nothing while - expansion is off. The base class checks the web first, so one pass would have dropped - that and then been refused about the rung; the loop gives up both and says so in one - line. This is the case that makes settling a loop rather than a single `replace`. - """ - from hmz.coganchor.agents import AntigravityCLIAgentConfig - - one = AntigravityCLIAgent( - AntigravityCLIAgentConfig( - model="a-model", effort="", disable_slash_commands=True - ) - ) - dropped: list[str] = [] - runs_at( - "bench.py", - one, - place(web_search=False, permission="read-only", insist=False), - dropped=dropped, - ) - assert one.config.web_search is not False - assert one.config.permission == UNSAID - assert len(dropped) == 1 - assert "permission" in dropped[0] - assert "web_search" in dropped[0] - - -class TestOnlyWhatThePlaceDeclared: - """Leniency gives up the flow's own answers and nobody else's. - - A tier and an effort are refused where the agent is built, which is upstream of every - flow, so neither of them can reach `runs_at` by way of a place at all -- and that is the - point rather than a gap: they are the settings whoever chose the agent chose, and there - is no path by which a flow gives one of them up. What does reach here is a refusal naming - a field beside one the place declared, and that is refused whatever the place says. - """ - - def test_a_refusal_naming_a_setting_the_place_did_not_declare_is_still_a_refusal( - self, - ) -> None: - """The guard itself, put to a backend that refuses something nobody here asked for. - - No backend in the package can be made to do this today -- a tier and an effort are - refused at construction, upstream of every flow -- which is why it is asked of a - stand-in rather than of one of them. The rule has to hold for the backend somebody - adds next: a refusal about a field the flow never spoke about is not the flow's to - wave away, however little it insisted about the field it did speak about. - """ - - class RefusesTheTier(PiAgent): - """Made happily, and refuses the tier the moment it is set up as anything.""" - - built = False - - def _serves(self, config: AgentConfig) -> None: - if not self.built: - return - raise Unserved("this one cannot be served fast", "service_tier") - - one = RefusesTheTier(AgentConfig(model="a-model", effort="")) - one.built = True - with pytest.raises(NotAFlow, match="cannot be served fast"): - runs_at("bench.py", one, place(web_search=False, insist=False)) - - def test_a_tier_no_backend_can_send_is_refused_where_the_agent_is_built( - self, - ) -> None: - """Upstream of every flow, which is why no place may give it up.""" - with pytest.raises(Unserved) as refused: - PiAgent(AgentConfig(model="a-model", effort="", service_tier="fast")) - assert refused.value.settings == frozenset({"service_tier"}) - - def test_an_effort_off_the_ladder_is_refused_where_the_agent_is_built(self) -> None: - """The same, for the rung it thinks at: a place never declares one.""" - with pytest.raises(Unserved) as refused: - ClaudeCodeAgent(AgentConfig(model="a-model", effort="ultrathink")) - assert refused.value.settings == frozenset({"effort"}) - - def test_the_three_a_place_may_declare_are_the_three_it_may_give_up(self) -> None: - """The rule underneath both, read on its own: silence declares nothing.""" - assert _declared(place()) == frozenset() - assert _declared(place(web_search=False)) == frozenset({"web_search"}) - assert _declared(place(permission="read-only")) == frozenset({"permission"}) - assert _declared(place(goals=False)) == frozenset({"goals"}) - assert "service_tier" not in _declared( - place(web_search=False, permission="read-only", goals=False) - ) - - -class TestTheRefusalNamesItself: - """What makes any of this possible: a shortfall that says which setting it was.""" - - def test_it_is_a_value_error_still(self) -> None: - """Every `except ValueError` written before this catches it now.""" - assert issubclass(Unserved, ValueError) - - def test_the_sentence_is_the_sentence_and_the_name_travels_beside_it(self) -> None: - """The message a person reads is unchanged; the field name is the new part.""" - one = Unserved("it cannot be told", "web_search") - assert str(one) == "it cannot be told" - assert one.settings == frozenset({"web_search"}) - - def test_a_backend_that_cannot_be_told_names_the_field(self) -> None: - """Raised where it always was, and now answerable about what it was about.""" - with pytest.raises(Unserved) as refused: - PiAgent(AgentConfig(model="a-model", effort="", web_search=False)) - assert refused.value.settings == frozenset({"web_search"}) - - def test_a_pair_refused_together_names_both(self) -> None: - """Opencode withholding its table hears neither, and dropping one fixes neither.""" - from hmz.coganchor.agents import OpencodeAgentConfig - - with pytest.raises(Unserved) as refused: - OpencodeAgentConfig( - model="anthropic/claude", - effort="", - permission="read-only", - web_search=False, - permission_table=False, - ) - assert refused.value.settings == frozenset({"permission", "web_search"}) - - def test_a_withheld_table_says_nothing_about_a_web_nobody_asked_about(self) -> None: - """`web_search=None` is not a narrowing, so it is not a thing to be refused. - - The three-answer switch has a silence in it, and a silence withholds nothing: a - config that never raised the subject asks this backend's table for nothing, and the - table not being written is no loss to it. `not self.web_search` read that silence as - a no the moment the field's default became None, which would have refused a config - that had asked for nothing at all. - """ - from hmz.coganchor.agents import OpencodeAgentConfig - - one = OpencodeAgentConfig( - model="anthropic/claude", - effort="", - permission="bypass", - permission_table=False, - ) - assert one.web_search is None - - -class TestTheDefaultDidNotMove: - """Nothing changes for a flow that does not ask, which is every flow there is.""" - - def test_a_place_that_says_nothing_insists(self) -> None: - assert AgentDefaults().insist is True - assert place().insist is True - - def test_a_place_declaring_nothing_still_settles_nothing(self) -> None: - """Leniency is about refusals, and a place with nothing to declare meets none.""" - one = agent(PiAgent) - was = one.config - dropped: list[str] = [] - assert runs_at("bench.py", one, place(insist=False), dropped=dropped) == was - assert one.config == was - assert dropped == [] - - -#: The same flow twice, once insisting and once not -- a place declaring that its answers -#: have to be the same tomorrow, which is the declaration flowbench makes and the one every -#: backend with no web switch used to fail. -INSISTS = '''"""A flow whose answers have to be the same tomorrow.""" - -from typing import Annotated - -from hmz.coganchor.agents import AgentBase, AgentDefaults -from hmz._legacy_flows import flow - - -@flow -def run( - agents: tuple[Annotated[AgentBase, AgentDefaults(web_search=False)]], task: str -) -> None: - agents[0](task) -''' - -LENIENT = INSISTS.replace( - "AgentDefaults(web_search=False)", "AgentDefaults(web_search=False, insist=False)" -) - - -class TestTheWholePath: - """From the annotation a flow is written with to the line whoever has a screen says. - - The pieces are tested apart above; this is the one test that walks all of them at once, - because a field that reaches `Place` but never reaches `Runner` would pass every one of - those and still leave a flow declaring into thin air. - """ - - def test_a_lenient_flow_loads_and_the_runner_says_what_it_gave_up( - self, tmp_path: Path - ) -> None: - where = tmp_path / "quiet.py" - where.write_text(LENIENT) - one = agent(PiAgent) - - runner = Runner(str(where), [one]) - - assert one.config.web_search is not False - said = runner.unserved() - assert "web_search" in said - assert "pi" in said - - def test_an_insisting_flow_is_refused_where_it_always_was( - self, tmp_path: Path - ) -> None: - where = tmp_path / "quiet.py" - where.write_text(INSISTS) - - with pytest.raises( - NotAFlow, match="no way of being told not to search the web" - ): - Runner(str(where), [agent(PiAgent)]) - - def test_a_run_that_carried_everything_says_nothing(self, tmp_path: Path) -> None: - """`unserved()` is "" for every flow that did not ask, which is every flow there is.""" - where = tmp_path / "quiet.py" - where.write_text(LENIENT) - - assert Runner(str(where), [agent(ClaudeCodeAgent)]).unserved() == "" diff --git a/tests/unit/flows/test_prophesying.py b/tests/unit/flows/test_prophesying.py deleted file mode 100644 index dc584c5c..00000000 --- a/tests/unit/flows/test_prophesying.py +++ /dev/null @@ -1,660 +0,0 @@ -"""Compiling an atlas: the narrower Python it is written in, and the prophecy it becomes. - -An ordinary flow is read by running it, and what it will do is nobody's to ask. An atlas -answers that before anything runs: its body is a declaration, this is the reading that holds -it to the subset a declaration is written in, and what comes out is a graph -- nodes, edges, -and the shapes that flow along them. Everything here is refused or compiled without importing -a line of it. -""" - -from __future__ import annotations - -import json -from typing import TYPE_CHECKING - -import pytest - -from hmz.runtime.flowing import PROPHECY, canonical, checked, digest, kept -from hmz.runtime.flowing.prophesying import Prophesied, prophesied -from tests.stubs import written - -if TYPE_CHECKING: - from pathlib import Path - -#: What every atlas below is written against: who it drives, what flows between its nodes, -#: and the two nodes themselves. The bodies are what differ, which is what is on trial. -HEAD = '''"""An atlas, for the reading to have something to read.""" - -from typing import NamedTuple - -from pydantic import BaseModel, Field - -from hmz._legacy_flows import Agent, atlas, logic, mind - - -class Agents(NamedTuple): - """Who it drives.""" - - writer: Agent - - -class Draft(BaseModel): - """What the writer produced.""" - - model_config = {"extra": "forbid"} - - text: str = Field(description="the draft") - - -class Verdict(BaseModel): - """What was made of it.""" - - model_config = {"extra": "forbid"} - - done: bool = Field(description="whether it is finished") - - -@mind -def write(agent: Agent, task: str) -> Draft: - """One turn of writing.""" - return agent(task, schema=Draft) - - -@logic -def judge(said: Draft) -> Verdict: - """Reads the draft, which is what a branch hangs off.""" - return Verdict(done=bool(said.text)) - - -@logic(rerun=False) -def stamp(said: Draft) -> None: - """A node a run picked up again steps past.""" - - -''' - -#: The straight one: a turn, a reading of it, and a loop that ends when the reading says so. -LOOP = '''@atlas -def run(agents: Agents, task: str) -> None: - """Writes until the reading of it says it is done.""" - draft = write(agents.writer, task) - verdict = judge(draft) - while not verdict.done: - draft = write(agents.writer, task) -''' - - -def _held(under: Path, body: str, name: str = "one") -> Prophesied: - """Writes one atlas out and compiles it. - - Args: - under: Where the flows are kept. - body: The body under :data:`HEAD`. - name: What to call the flow. - - Returns: - What compiling it came to. - """ - return prophesied(written(under, name, HEAD + body)) - - -def _codes(under: Path, body: str) -> list[str]: - """Every error one atlas's body is refused for, by code.""" - held = _held(under, body) - assert held.prophecy is None - return [one.code for one in held.findings if one.severity == "error"] - - -def test_a_body_compiles_to_the_graph_it_declares(tmp_path: Path) -> None: - """One node per call, one edge per way from one to the next, and nothing run.""" - held = _held(tmp_path, LOOP) - - assert held.findings == () - prophecy = held.prophecy - assert prophecy is not None - assert [one.at for one in prophecy.nodes] == ["write", "judge", "write:2"] - assert [one.kind for one in prophecy.nodes] == ["mind", "logic", "mind"] - assert prophecy.agents == ("writer",) - assert prophecy.takes == "str" - - -def test_a_loop_is_an_edge_back_to_the_node_the_branch_reads(tmp_path: Path) -> None: - """Which is what makes the head answer again with whatever the round changed.""" - prophecy = _held(tmp_path, LOOP).prophecy - assert prophecy is not None - - ways = {(one.out_of, one.into): one.when for one in prophecy.edges} - assert ways[("", "write")] is None # the way in - assert ways[("write", "judge")] is None - rounds = ways[("judge", "write:2")] - assert rounds is not None - assert rounds.truth is False # round again while the reading says it is not done - assert ways[("write:2", "judge")] is None # and back to the node that reads it - over = ways[("judge", "")] - assert over is not None - assert over.truth is True # and out of the graph when it is - - -def test_the_same_body_written_twice_compiles_to_the_same_bytes(tmp_path: Path) -> None: - """Canonical means what it says: a comment is not part of what the atlas is.""" - one = _held(tmp_path, LOOP, "one").prophecy - two = _held( - tmp_path, - LOOP.replace( - " draft = write", " # a comment nobody compiles\n draft = write", 1 - ), - "two", - ).prophecy - assert one is not None - assert two is not None - - assert canonical(one) == canonical(two._replace(name="one")) - assert digest(one) == digest(two._replace(name="one")) - - -def test_what_a_node_answers_with_is_written_down_field_by_field( - tmp_path: Path, -) -> None: - """A shape is what an edge carries, and what carries it is what both ends are held to.""" - prophecy = _held(tmp_path, LOOP).prophecy - assert prophecy is not None - - said = json.loads(canonical(prophecy)) - assert {one["name"] for one in said["shapes"]} == {"Draft", "Verdict", "str"} - verdict = next(one for one in said["shapes"] if one["name"] == "Verdict") - assert verdict["fields"] == [["done", "bool", True]] - - -@pytest.mark.parametrize( - ("why", "body", "code"), - [ - ( - "a turn has one way out", - '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - if draft.text: - draft = write(agents.writer, task) -''', - "branching-mind", - ), - ( - "work is what a node is for", - '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - said = draft.text + "!" -''', - "unstatic-body", - ), - ( - "what flows in is what the far end takes", - '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - again = write(agents.writer, draft) -''', - "shape-mismatch", - ), - ( - "an agent it does not drive", - '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.reviewer, task) -''', - "unknown-agent", - ), - ( - "a loop nothing inside can end", - '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - verdict = judge(draft) - while not verdict.done: - pass -''', - "dead-loop", - ), - ( - "a node stepped past has no answer to leave behind", - '''@logic(rerun=False) -def marked(said: Draft) -> Verdict: - """Answers, and says it is stepped past.""" - return Verdict(done=True) - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - verdict = marked(draft) -''', - "skipped-answer", - ), - ( - "a name nothing bound", - '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - verdict = judge(nowhere) -''', - "unbound-read", - ), - ( - "a plain tuple says only how many", - '''@atlas -def run(agents: tuple[Agent], task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) -''', - "unnamed-agents", - ), - ( - "two decisions carried on one edge", - '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - verdict = judge(draft) - if verdict.done: - draft = write(agents.writer, task) - elif verdict.done: - draft = write(agents.writer, task) -''', - "unstatic-body", - ), - ( - "a name keeps the shape it was bound with", - '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - draft = judge(draft) -''', - "shape-mismatch", - ), - ( - "a graph with no nodes", - '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" -''', - "unstatic-body", - ), - ( - "a node that says nothing about what flows through it", - '''@logic -def loose(said): - """Says nothing.""" - return said - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - said = loose(task) -''', - "unshaped-node", - ), - ( - "a logic handed an agent", - '''@logic -def turned(agent: Agent, said: Draft) -> Verdict: - """Takes one, and is not a turn.""" - return Verdict(done=True) - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - verdict = turned(agents.writer, draft) -''', - "unagented-node", - ), - ], -) -def test_the_body_an_atlas_may_not_hold( - tmp_path: Path, why: str, body: str, code: str -) -> None: - """Every one of these is decidable, which is the bargain an atlas makes.""" - assert code in _codes(tmp_path, body), why - - -def test_an_atlas_reaches_an_atlas_and_nothing_else(tmp_path: Path) -> None: - """`load` answers with a flow that may be anything, which is a hole in a graph.""" - body = '''from hmz._legacy_flows import load - -chat = load("chat") - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) -''' - assert "dynamic-call" in _codes(tmp_path, body) - - -def test_a_supernode_is_the_atlas_under_it_compiled(tmp_path: Path) -> None: - """One node from outside, one prophecy from within.""" - body = '''@atlas(name="inner") -def inner(agents: Agents, said: Draft) -> Verdict: - """A whole atlas, reached as one node.""" - verdict = judge(said) - return verdict - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - verdict = inner(agents, draft) -''' - prophecy = _held(tmp_path, body).prophecy - assert prophecy is not None - - node = prophecy.node("inner") - assert node is not None - assert node.kind == "atlas" - assert node.under == "inner" - under = prophecy.under("inner") - assert under is not None - assert [one.at for one in under.nodes] == ["judge"] - assert under.takes == "Draft" - assert under.gives == "Verdict" - - -def test_a_supernode_that_reaches_back_into_its_own_graph_is_refused( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - """A graph inside a graph inside itself has no bottom, however it is spelled. - - Reached by name here rather than beside it, since the two spellings of one atlas are - exactly what a check comparing names would follow forever. - """ - monkeypatch.setenv("HOME", str(tmp_path / "home")) - monkeypatch.chdir(tmp_path) - body = '''from hmz._legacy_flows import sub - -again = sub("one:inner") - - -@atlas(name="inner") -def inner(agents: Agents, said: Draft) -> Verdict: - """Reaches back into itself.""" - verdict = again(agents, said) - return verdict - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - verdict = inner(agents, draft) -''' - assert "circular-atlas" in _codes(tmp_path / ".humanize/flows", body) - - -def test_a_flow_that_is_not_an_atlas_is_not_compiled(tmp_path: Path) -> None: - """`checked` is the reading for those, and it is what it goes on being.""" - plain = '''"""An ordinary flow.""" - -from hmz._legacy_flows import Agent, flow - - -@flow -def run(agents: tuple[Agent], task: str) -> None: - """Does whatever it likes.""" -''' - at = written(tmp_path, "plain", plain) - - assert prophesied(at).prophecy is None - assert [one.code for one in prophesied(at).findings] == ["not-an-atlas"] - assert checked(at) == () # and the ordinary reading has nothing against it - - -def test_the_stricter_reading_is_the_one_an_atlas_gets(tmp_path: Path) -> None: - """Checking asks one question, and an atlas is what decides which reading answers.""" - body = '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - if draft.text: - draft = write(agents.writer, task) -''' - at = written(tmp_path, "one", HEAD + body) - - assert "branching-mind" in {one.code for one in checked(at)} - - -def test_a_shipped_prophecy_that_is_no_longer_the_source_is_said( - tmp_path: Path, -) -> None: - """A run walks the shipped one, so a drifted flow reads as something it is not.""" - at = written(tmp_path, "one", HEAD + LOOP) - held = prophesied(at).prophecy - assert held is not None - (at / PROPHECY).write_bytes(kept(held._replace(gives="Draft"))) - - assert "stale-prophecy" in {one.code for one in checked(at)} - - -def test_a_prophecy_that_cannot_be_read_back_is_not_walked(tmp_path: Path) -> None: - """Bytes that are not a prophecy are a file to compile again, not a graph to guess at.""" - at = written(tmp_path, "one", HEAD + LOOP) - (at / PROPHECY).write_bytes(b"nothing here is a prophecy") - - assert "stale-prophecy" in {one.code for one in checked(at)} - - -def test_a_supernode_that_says_it_can_be_set_up_is_refused(tmp_path: Path) -> None: - """What is set up is the run, so such an atlas is one to start and not one to reach.""" - body = '''class Config(BaseModel): - """What it takes.""" - - model_config = {"extra": "forbid"} - - rounds: int = Field(default=2, description="how many rounds it takes") - - -@atlas(name="inner") -def inner(agents: Agents, said: Draft, config: Config | None = None) -> Verdict: - """A graph that says it can be set up.""" - verdict = judge(said) - return verdict - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - verdict = inner(agents, draft) -''' - assert "unstatic-body" in _codes(tmp_path, body) - - -def test_what_an_atlas_can_be_set_up_with_is_part_of_the_prophecy( - tmp_path: Path, -) -> None: - """A node may read the config, so what it is is what the graph is checked against.""" - body = '''class Config(BaseModel): - """What it takes.""" - - model_config = {"extra": "forbid"} - - rounds: int = Field(default=2, description="how many rounds it takes") - - -@logic -def bounded(said: Draft, rounds: int) -> Verdict: - """Reads the draft against the bound the run was set up with.""" - return Verdict(done=bool(said.text) and rounds > 0) - - -@atlas -def run(agents: Agents, task: str, config: Config | None = None) -> None: - """Says it.""" - draft = write(agents.writer, task) - verdict = bounded(draft, config.rounds) -''' - prophecy = _held(tmp_path, body).prophecy - assert prophecy is not None - - assert prophecy.config == "Config" - node = prophecy.node("bounded") - assert node is not None - assert node.takes[1].reads == "@config" - assert node.takes[1].field == "rounds" - - -def test_a_way_out_carries_what_the_atlas_answers_with(tmp_path: Path) -> None: - """Which is what the `return` named, and not whatever the last node happened to say.""" - body = '''@atlas(name="inner") -def inner(agents: Agents, said: Draft) -> Draft: - """Answers with something it bound two nodes ago.""" - draft = write(agents.writer, said.text) - verdict = judge(draft) - return draft - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - again = inner(agents, draft) -''' - prophecy = _held(tmp_path, body).prophecy - assert prophecy is not None - under = prophecy.under("inner") - assert under is not None - - out = next(one for one in under.edges if not one.into) - assert out.answers == "draft" - - -def test_an_atlas_that_answers_says_so_on_every_way_out(tmp_path: Path) -> None: - """A path running off the bottom of the body answers with nothing, which is not it.""" - body = '''@atlas(name="inner") -def inner(agents: Agents, said: Draft) -> Draft: - """Says it answers with a draft and runs off the bottom instead.""" - draft = write(agents.writer, said.text) - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - again = inner(agents, draft) -''' - assert "shape-mismatch" in _codes(tmp_path, body) - - -def test_the_atlas_a_name_asks_for_is_the_one_that_is_read(tmp_path: Path) -> None: - """A file may hold several, and `official/review:pass` is which of them was meant.""" - body = '''@atlas(name="pass") -def only(agents: Agents, task: str) -> None: - """The only atlas the file holds, and it has a name of its own.""" - draft = write(agents.writer, task) -''' - at = written(tmp_path, "one", HEAD + body) - - assert prophesied(at, name="pass").prophecy is not None - # And the file's own name asks for one it does not hold, which it says and lists. - said = prophesied(at).findings - assert [one.code for one in said] == ["not-an-atlas"] - assert "'pass'" in said[0].said - - -def test_python_beside_an_atlas_is_read_as_python(tmp_path: Path) -> None: - """The bodies left out are the ones compiled, and not everything spelled like one.""" - body = '''class Helper: - """Something beside the atlas with a method the atlas's name also uses.""" - - def run(self) -> None: - """A loop nothing inside can end, which is a thing to be told about.""" - while True: - pass - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) -''' - at = written(tmp_path, "one", HEAD + body) - - assert "sleeping-loop" in {one.code for one in checked(at)} - - -def test_a_node_the_walk_cannot_await_is_refused(tmp_path: Path) -> None: - """A coroutine bound as an answer is what the next node cannot be built from.""" - body = '''@mind -async def slowly(agent: Agent, task: str) -> Draft: - """A node that is a coroutine.""" - return Draft(text=task) - - -@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = slowly(agents.writer, task) -''' - assert "unstatic-body" in _codes(tmp_path, body) - - -def test_a_config_a_run_may_not_be_set_up_with_has_defaults(tmp_path: Path) -> None: - """A body cannot write `config or Config()`, so the model has to stand in for itself.""" - body = '''class Held(BaseModel): - """What it takes, and cannot be built without.""" - - model_config = {"extra": "forbid"} - - rounds: int = Field(description="how many rounds, with nothing to fall back on") - - -@atlas -def run(agents: Agents, task: str, config: Held | None = None) -> None: - """Says it.""" - draft = write(agents.writer, task) -''' - assert "unset-config" in _codes(tmp_path, body) - - -def test_a_loop_body_that_ends_with_the_node_it_reads_is_refused( - tmp_path: Path, -) -> None: - """The edge back runs the head again, so writing it twice is running it twice.""" - body = '''@atlas -def run(agents: Agents, task: str) -> None: - """Says it.""" - draft = write(agents.writer, task) - verdict = judge(draft) - while not verdict.done: - draft = write(agents.writer, task) - verdict = judge(draft) -''' - assert "twice-round" in _codes(tmp_path, body) - - -def test_one_thing_wrong_in_a_body_is_one_finding(tmp_path: Path) -> None: - """A name a refused statement would have bound is that mistake again, not another.""" - body = '''@atlas -def run(agents: Agents, task: str) -> None: - """A body the subset does not allow.""" - for one in [1, 2]: - draft = write(agents.writer, task) - verdict = judge(draft) - while not verdict.done: - draft = write(agents.writer, task) -''' - said = _codes(tmp_path, body) - - # The `for`, and nothing about the names it would have bound, nor about the body - # having no nodes in it -- both of which follow from it rather than stand beside it. - assert said == ["unstatic-body"] diff --git a/tests/unit/flows/test_settling_ladder.py b/tests/unit/flows/test_settling_ladder.py deleted file mode 100644 index 463ad5c1..00000000 --- a/tests/unit/flows/test_settling_ladder.py +++ /dev/null @@ -1,116 +0,0 @@ -"""The rule a place is settled onto an agent by, read on its own, apart from any flow. - -Settling is one line of :func:`hmz.runtime.flowing.driving.runs_at` and the whole of what a flow's -declaration comes to: two answers go in -- what the agent already carries and what the place -declares -- and the narrower of them comes out, so that a flow calling one somebody else wrote -cannot have its `read-only` undone by a callee that declared nothing. - -What is new here is that a config may carry no answer at all. `UNSAID` is not a rung on the -ladder but the absence of one: humanize saying nothing to the CLI about what its agent may do, -which leaves the agent wherever that CLI's own headless run leaves it. It has to sit looser -than every rung, or the tighten-only rule would read silence as a tightening and a place -declaring nothing would start settling something. - -Tested here rather than through a flow because there is no agent, no process and no CLI in any -of it: the ladder is a tuple and the rule is a `min` over it, and what a reader wants to know -is which answer wins. -""" - -from __future__ import annotations - -import pytest - -from hmz.coganchor.agents import ( - PERMISSIONS, - UNSAID, - AgentConfig, - AgentDefaults, - searching, - tightest, -) - - -def test_silence_on_both_sides_stays_silence() -> None: - """Two configs that say nothing settle to saying nothing, which is the whole point. - - This is the case the work is for: a run given no special parameters, filling a place that - declared no rung, has to reach the CLI as bare as a person launching it headless by hand. - """ - assert tightest(UNSAID, UNSAID) == UNSAID - - -@pytest.mark.parametrize("rung", PERMISSIONS) -def test_any_rung_beats_the_silence_above_it(rung: str) -> None: - """A declared rung settles onto silence, and silence never settles onto a rung. - - Both directions, because a place and an agent are the two sides of the same call: a flow - declaring `read-only` over an agent that said nothing gets `read-only`, and an agent - carrying `read-only` under a place that said nothing keeps it. - """ - assert tightest(UNSAID, rung) == rung - assert tightest(rung, UNSAID) == rung - - -def test_the_tighter_of_two_rungs_wins() -> None: - """The rule the ladder was written for, unchanged by the silence sitting above it.""" - assert tightest("read-only", "bypass") == "read-only" - assert tightest("bypass", "read-only") == "read-only" - assert tightest("workspace-write", "auto") == "workspace-write" - - -@pytest.mark.parametrize("was", [*PERMISSIONS, UNSAID]) -@pytest.mark.parametrize("said", [*PERMISSIONS, UNSAID]) -def test_which_side_an_answer_was_written_on_makes_no_difference( - was: str, said: str -) -> None: - """Symmetric in its arguments, over every pair there is. - - Nothing reads the argument names to decide: settling is the narrower of two answers, and a - rule that came out differently depending on which of them was the agent's would make a - flow's behaviour depend on how its caller happened to be configured. - """ - assert tightest(was, said) == tightest(said, was) - - -@pytest.mark.parametrize( - ("was", "said", "settled"), - [ - (None, None, None), - (None, True, True), - (True, None, True), - (True, True, True), - (True, False, False), - (False, None, False), - (False, True, False), - (None, False, False), - ], -) -def test_the_web_is_read_only_where_nobody_withheld_it( - was: bool | None, said: bool | None, settled: bool | None -) -> None: - """Off beats on, and on beats unsaid: the same ladder, two rungs and a silence. - - `None` is the run that never mentions searching, so the CLI is never told either way and - goes on doing whatever it does unasked. An explicit `True` is humanize saying so, which is - a narrowing of that silence rather than a loosening: a backend that cannot be told refuses - it, and refusing is what a setting that would otherwise lie is owed. - """ - assert searching(was, said) is settled - - -def test_a_config_may_be_written_with_no_rung_at_all() -> None: - """Both places a permission is written accept the silence, because it is an answer.""" - assert AgentConfig(model="m", effort="", permission=UNSAID).permission == UNSAID - assert AgentDefaults(permission=UNSAID).permission == UNSAID - - -def test_a_rung_no_backend_has_a_word_for_is_still_refused() -> None: - """Widening the answers is not opening them: a typo is refused where it is written. - - Both places, and with the sentence it always had: what a reader of the refusal needs is - the rungs there are rather than the silence, which is what they get by writing nothing. - """ - with pytest.raises(ValueError, match="permission must be one of"): - AgentConfig(model="m", effort="", permission="whatever") - with pytest.raises(ValueError, match="permission must be one of"): - AgentDefaults(permission="whatever") diff --git a/tests/unit/runtime/test_kept.py b/tests/unit/runtime/test_kept.py index 754f30d9..7e96802a 100644 --- a/tests/unit/runtime/test_kept.py +++ b/tests/unit/runtime/test_kept.py @@ -1,50 +1,140 @@ -"""An agent as it goes into a file and comes back out of one. +"""An agent as it goes into a file and comes back out of one, and what a flow is set up with. -Three answers apiece and two of them may be withheld: what an agent may do without being -asked, and whether it may search the web, are things a place may simply not have said. A file -that turned either silence into an answer would be the interface deciding, on somebody's -behalf and without saying so, what every agent it wrote down is allowed. +An agent is a CLI, an account, and a model at an effort: the word `-a` takes after its role, +and nothing else -- what it may do and where it works are the flow's to say. And what a +workspace remembers of a flow is what each of its roles was given, its params and its budget, +each left alone where it is not handed in again and erased where it is handed in empty. """ from __future__ import annotations -from hmz.runtime.kept import Runs, read_back, written - - -def test_an_agent_nobody_narrowed_is_written_down_as_the_silence_it_is() -> None: - """A rung nobody chose is a key that is not there, and comes back as no rung at all.""" - runs = Runs("claude/m:high") - - held = written(runs) - - assert "permission" not in held - # And web search is written even so, because here the silence has to be told from a file - # older than the question: a key that is there and empty is this run saying nobody was - # asked, and no key at all is a file written before anybody could be. - assert held["web_search"] is None - - back = read_back(held) - - assert back == runs - assert back is not None - assert back.permission == "" +from typing import TYPE_CHECKING +import pytest -def test_an_answer_somebody_gave_is_written_down_and_comes_back_as_itself() -> None: - """Which is the whole of what the silence beside it is for: the two are different.""" - runs = Runs("claude/m:high", permission="read-only", web_search=False) - - held = written(runs) - - assert held["permission"] == "read-only" - assert held["web_search"] is False - assert read_back(held) == runs - - -def test_a_file_older_than_the_question_reads_as_what_every_agent_did_then() -> None: - """Every agent of every flow searched the web, and none of them was ever asked.""" - older = read_back({"cli": "claude", "model": "m", "effort": "high"}) - - assert older is not None - assert older.web_search is True - assert older.permission == "" +from hmz.runtime.kept import Runs, read_back, written +from hmz.runtime.settings import Settings + +if TYPE_CHECKING: + from pathlib import Path + + +def test_an_agent_is_written_down_as_the_word_a_line_takes() -> None: + assert written(Runs("claude/m:high")) == "claude/m:high" + assert written(Runs("claude/m:high", "work")) == "claude@work/m:high" + + +@pytest.mark.parametrize( + "runs", + [ + Runs("claude/m:high"), + Runs("claude/m:high", "work"), + # A model's own punctuation stays the model's: read from both ends. + Runs("opencode/openrouter/some:model:low"), + Runs("codex/gpt-5.5:auto", "a@b"), + ], +) +def test_an_agent_written_down_comes_back_as_itself(runs: Runs) -> None: + assert read_back(written(runs)) == runs + + +@pytest.mark.parametrize( + "held", + [ + None, + {"cli": "claude", "model": "m", "effort": "high"}, # an older humanize's + "claude", + "claude/m", + "/m:high", + "claude/:high", + ], +) +def test_what_is_not_one_reads_back_as_nothing(held: object) -> None: + assert read_back(held) is None + + +def test_what_a_flow_was_set_up_with_is_read_back_by_role(tmp_path: Path) -> None: + Settings(tmp_path).remember( + "rlar", + {"builder": Runs("claude/m:high"), "reviewer": Runs("codex/n:low", "work")}, + envs={"remote": "ssh@box/home/me/repo"}, + params={"rounds": 3}, + budget={"cost": 5.0}, + ) + + held = Settings(tmp_path) + assert held.flow == "rlar" + assert held.agents("rlar") == { + "builder": Runs("claude/m:high"), + "reviewer": Runs("codex/n:low", "work"), + } + assert held.envs("rlar") == {"remote": "ssh@box/home/me/repo"} + assert held.params("rlar") == {"rounds": 3} + assert held.budget("rlar") == {"cost": 5.0} + assert held.agents("chat") == {} + + +def test_choosing_the_agents_again_leaves_the_rest_alone_and_empty_erases( + tmp_path: Path, +) -> None: + settings = Settings(tmp_path) + settings.remember( + "rlar", + {"builder": Runs("claude/m:high")}, + envs={"remote": "ssh@box/repo"}, + params={"rounds": 3}, + budget={"cost": 5.0}, + ) + + settings.remember("rlar", {"builder": Runs("codex/n:low")}) + held = Settings(tmp_path) + assert held.agents("rlar") == {"builder": Runs("codex/n:low")} + assert held.envs("rlar") == {"remote": "ssh@box/repo"} + assert held.params("rlar") == {"rounds": 3} + assert held.budget("rlar") == {"cost": 5.0} + + settings.remember("rlar", {"builder": Runs("codex/n:low")}, params={}, budget={}) + held = Settings(tmp_path) + assert held.params("rlar") == {} + assert held.budget("rlar") == {} + assert held.envs("rlar") == {"remote": "ssh@box/repo"} + + +def test_what_humanize_did_not_write_reads_as_nothing_remembered( + tmp_path: Path, +) -> None: + """An older humanize wrote an agent as a mapping of its fields; that is not one now.""" + import yaml + + from hmz import home + + at = home() / "settings.yaml" + at.parent.mkdir(parents=True, exist_ok=True) + at.write_text( + yaml.safe_dump( + { + "workspaces": { + str(tmp_path.resolve()): { + "flow": "rlar", + "flows": { + "rlar": { + "agents": { + "builder": { + "cli": "claude", + "model": "m", + "effort": "high", + } + }, + "envs": {"remote": 3}, + } + }, + } + } + } + ) + ) + + held = Settings(tmp_path) + assert held.flow == "rlar" + assert held.agents("rlar") == {} + assert held.envs("rlar") == {} diff --git a/tests/unit/runtime/test_telemetry.py b/tests/unit/runtime/test_telemetry.py index 4db37009..22c290e0 100644 --- a/tests/unit/runtime/test_telemetry.py +++ b/tests/unit/runtime/test_telemetry.py @@ -74,8 +74,8 @@ def test_a_workspace_forgotten_is_forgotten_whichever_one_did_it( """A merge cannot see an absence: only what this instance read when it opened tells it.""" from hmz.runtime.kept import Runs - Settings(tmp_path / "one").remember("chat", ("a",), [Runs("claude/m:high")]) - Settings(tmp_path / "other").remember("rlar", ("a",), [Runs("codex/n:low")]) + Settings(tmp_path / "one").remember("chat", {"a": Runs("claude/m:high")}) + Settings(tmp_path / "other").remember("rlar", {"a": Runs("codex/n:low")}) assert Settings(tmp_path / "one").forget(str((tmp_path / "other").resolve())) @@ -332,7 +332,7 @@ def test_a_setting_written_elsewhere_survives_a_workspace_being_remembered( one, other = Settings(tmp_path), Settings(tmp_path) one.answers(enable_sentry=True) - other.remember("chat", ("a",), [Runs("claude/m:high")]) + other.remember("chat", {"a": Runs("claude/m:high")}) read = Settings(tmp_path) assert read.enable_sentry is True @@ -347,9 +347,9 @@ def test_a_workspace_may_be_forgotten_without_forgetting_anything_else( Settings(tmp_path).answers(enable_sentry=False) kept = Settings(tmp_path) - kept.remember("chat", ("a",), [Runs("claude/m:high")]) + kept.remember("chat", {"a": Runs("claude/m:high")}) elsewhere = Settings(tmp_path / "other") - elsewhere.remember("rlar", ("a",), [Runs("codex/n:low")]) + elsewhere.remember("rlar", {"a": Runs("codex/n:low")}) assert Settings(tmp_path).forget() diff --git a/tests/unit/tui/test_skills.py b/tests/unit/tui/test_skills.py index a8613075..3111eab1 100644 --- a/tests/unit/tui/test_skills.py +++ b/tests/unit/tui/test_skills.py @@ -275,22 +275,24 @@ def test_a_skill_with_no_front_matter_is_the_directory_it_is_in(homes: Path) -> def test_a_workspace_writes_down_no_skills_of_its_own(tmp_path: Path) -> None: """What an agent is does not include them any more, so nothing about them is kept.""" kept = Settings(tmp_path) - kept.remember("rlar", ("actor",), [Runs("claude/m:high")]) + kept.remember("rlar", {"actor": Runs("claude/m:high")}) - assert Settings(tmp_path).agents("rlar") == [Runs("claude/m:high")] + assert Settings(tmp_path).agents("rlar") == {"actor": Runs("claude/m:high")} held = Settings(tmp_path)._read() agents = held["workspaces"][str(tmp_path.resolve())]["flows"]["rlar"]["agents"] - assert "skills" not in agents["actor"] + assert agents["actor"] == "claude/m:high" # the word `-a` takes, and nothing else -def test_a_file_that_still_says_skills_is_read_past(tmp_path: Path) -> None: - """An agent written down when they were a setting is the agent it always was.""" +def test_a_file_that_still_says_skills_is_read_as_nothing_remembered() -> None: + """An agent written down in the shape it had when skills were a setting is not one now. + + What an agent is written down as is the word `-a` takes, and a file holding anything + else is one humanize did not write this way: it reads as nothing remembered, which is a + flow asked about again rather than one started on half an answer. + """ from hmz.runtime.kept import read_back - runs = read_back( - {"cli": "claude", "model": "m", "effort": "high", "skills": ["writing"]} + assert ( + read_back({"cli": "claude", "model": "m", "effort": "high", "skills": ["a"]}) + is None ) - - # Searching the web among them: a file this old was written before there was such a - # setting, and every agent of every flow then searched. - assert runs == Runs("claude/m:high", web_search=True)