From 33fc9122b1e1a2170b9cc467efba4895467c6521 Mon Sep 17 00:00:00 2001 From: Joe LaPenna Date: Fri, 2 Oct 2026 04:33:36 +0000 Subject: [PATCH 1/2] docs: make agent context retrieval-oriented Refs #70 --- AGENTS.md | 198 +++++++------------------------ ARCHITECTURE.md | 77 ++++++++++++ README.md | 3 + docs/README.md | 20 ++++ tests/unit/test_agent_context.py | 62 ++++++++++ 5 files changed, 207 insertions(+), 153 deletions(-) create mode 100644 ARCHITECTURE.md create mode 100644 docs/README.md create mode 100644 tests/unit/test_agent_context.py diff --git a/AGENTS.md b/AGENTS.md index b1c50e86..e92a437b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,153 +1,45 @@ -# AGENTS.md - -For every new session, you **MUST** do the following: - -1. Before making a plan or writing code, ALWAYS use the `grep_search` tool on the `skills/` and `sparkrun/sparkrun-cc-plugin/skills/` directories for keywords related to the user's prompt to identify the correct protocol to follow. (Do NOT try to read all skills at once). -1. Read and strictly follow all rules defined in `.agents/rules/`. -1. Activate local (this repo only) skills: `stack-manager`, `source-dependency-dev`, `monitoring`, `stack-upkeep`, `stack-debugging` -1. When working with Python, Docker, or architecture, explicitly search for and load relevant skills like `python-pro`, `async-python-patterns`, `docker-expert`, and `senior-architect` to apply modern best practices. - -## Knowledge and memory. - -1. Do not add operational learnings, incident findings, or debugging discoveries to AGENTS.md or GEMINI.md. The static project documentation below is intentional — do not confuse it with accumulated knowledge. -1. Prefer creating or updating skills in the skills/ directory to keep track of learned information, processes or techniques. - -______________________________________________________________________ - -# Project Overview: Spark Services Orchestrator - -Spark Services Orchestrator (`sparkstack`) is a high-performance deployment orchestrator for the Spark ecosystem. It manages a suite of AI services, including the **OpenClaw** backend gateway, the **SparkRun** orchestrator, and various local LLM inference stacks (powered by **vLLM**). - -The project is designed for Linux hosts, utilizing Docker and Docker Compose for secure service isolation and networking. It focuses on robust, async-first orchestration scripts to manage the full lifecycle of the AI stack. - -## Core Technologies - -- **Language:** Python 3.13+ (Async-first) -- **Package Manager:** [uv](https://github.com/astral-sh/uv) -- **Containerization:** Docker & Docker Compose V2 -- **Key Services:** - - **OpenClaw:** Backend gateway and API router (Managed as a read-only Git source dependency in ../openclaw). - - **SparkRun:** Automated orchestration and evaluation (Managed as an editable path source dependency in ../sparkrun). - - **vLLM:** High-throughput LLM inference backend. -- **Monitoring:** Prometheus, Grafana, and Tempo (Managed via Grafana Alloy). -- **Registry:** `sparkstack-registry` for deployment recipes and model configurations. - -## Building and Running - -The project is a Python package (`sparkstack`) managed by `uv`. All operations use the unified `sparkstack` CLI. - -- **Initialization:** - ```bash - make setup # Synchronizes dependencies and installs pre-commit hooks - ``` -- **Deploying/Updating the Stack:** - ```bash - sparkstack update [SERVICES]... # Full service update orchestration (e.g. sparkstack update openclaw) - sparkstack update --json # Same, with JSON-Lines output (headless) - sparkstack set-current # Switch active stack and restart services - sparkstack launch # Launch services for a specific stack - sparkstack status # Live deployment monitor TUI (connects via UDS) - ``` -- **Building:** - ```bash - sparkstack build [--gpu-filter ] - ``` -- **Utilities:** - ```bash - sparkstack wait # Wait for backend models to load - sparkstack sync-registry # Sync model registry into OpenClaw - sparkstack update-monitoring # Apply monitoring stack updates - sparkstack check memory # Audit memory usage vs budget - sparkstack verify-model # Check HF model exists - sparkstack clear-sessions # Reset stuck OpenClaw sessions - ``` -- **Stack Verification:** - ```bash - uv run pytest tests/e2e/ # Run e2e tests only after you've updated services - ``` - -### Headless / Scripted Usage (`--json`) - -Most commands accept `--json` to emit structured **JSON-Lines** to stdout instead of -interactive Rich/TUI output. This is the **preferred interface for automated agents -and CI pipelines** — do NOT attempt to parse or navigate the interactive TUI -programmatically. - -**Commands supporting `--json`:** `update`, `build`, `wait`, `set-current`, `launch`, `check`, `sync-registry`. - -Each line is a self-contained JSON object: - -```json -{"event_type": "log", "level": "INFO", "message": "Deploying stack", "timestamp": "...", "service": "vLLM", "phase": "Restarting stack"} -``` - -**Usage patterns:** - -```bash -# Run a full update headlessly, pipe events to jq -sparkstack update --json | jq -r 'select(.level == "ERROR") | .message' - -# Wait for backends with structured progress -sparkstack wait --json - -# Build and capture result -sparkstack build my-recipe --json > build_events.jsonl -``` - -When `--json` is active: - -- Human-readable Rich progress bars and spinners are suppressed. -- All log events are written as JSON-Lines to **stdout**. -- Errors and debug logs are still written to `update_services.log`. -- The IPC UDS socket (`/tmp/sparkstack.sock`) still broadcasts events for any connected TUI clients. - -## Development Conventions - -### Source Dependency Policy - -- **OpenClaw (`../openclaw/`)**: Treated as a source dependency mainly maintained on the `local-dev` branch, which gets rebased against a tagged stable release. When working with OpenClaw, ensure you are on `local-dev` and follow a similar PR feature integration workflow as SparkRun if modifications are needed. -- **SparkRun (`../sparkrun/`)**: Treated as an editable source dependency. **CRITICAL:** You must ensure the `local-dev` branch is checked out before making any modifications or running tests. You **MUST** strictly follow the "Trunk-Based Feature Integration" workflow documented in the `source-dependency-dev` skill for any `sparkrun` changes (creating feature branches from `main`, merging them into `local-dev`, and rebasing `local-dev` via the orchestration script). - -### Scripting Standards - -- **Idiomatic Execution**: Use `sparkstack ` (or `uv run sparkstack `) for all operations. `package = true` in `pyproject.toml`. -- **Headless Execution**: Always use `--json` when calling `sparkstack` from scripts or automated agents. Never rely on parsing interactive terminal output. -- **Context Awareness**: Scripts should be domain-agnostic and use Pydantic schemas from `core/schemas.py`. - -### Planning and Verification - -- **Planning**: For tasks related to `stack-manager`, use the templates in `skills/stack-manager/references/plan-template.md`. -- **Verification**: Only run e2e tests (`uv run pytest tests/e2e/`) after you've run `update_services.py`. - -## IPC Monitoring Architecture - -`sparkstack update` embeds an IPC server broadcasting JSON-Lines events over a UNIX Domain Socket (`/tmp/sparkstack.sock`). There are two ways to consume these events: - -1. **Interactive TUI:** `sparkstack status` — a Textual app that connects to the UDS for live dashboard monitoring. -1. **Headless JSON:** `sparkstack update --json` — emits the same events as JSON-Lines to stdout, suitable for piping into `jq`, logging aggregators, or automated agents. - -``` -sparkstack update (Orchestrator + IPCServer) ──UDS──▶ sparkstack status (Textual TUI) - ──stdout──▶ --json (JSON-Lines for scripts) -``` - -- **Protocol:** Newline-delimited JSON over UDS. Event types: `state`, `full_sync`, `log`, `exit`. -- **Headless safe:** When no TUI client is connected, the orchestrator operates identically. -- **IPC Server:** `sparkstack/core/ipc_server.py` — async UDS server with per-client full_sync on connect. - -## Directory Structure - -- `sparkstack/cli/`: Unified Click CLI entry point (`sparkstack` command). -- `sparkstack/cli/_status.py`: Textual TUI client for live deployment monitoring. -- `sparkstack/core/`: Shared async utilities, health probes, and Pydantic configuration schemas. -- `sparkstack/core/ipc_server.py`: IPC server for UDS event broadcasting during updates. -- `sparkstack/manager/`: High-level orchestration scripts for building, updating, and syncing the stack. -- `services/`: Configuration fragments, `docker-compose.yml` files, and service-specific managers. -- `skills/`: Local AI agent skills (e.g., `stack-manager`, `source-dependency-dev`). -- `../sparkstack-registry/`: Source dependency containing model and stack deployment recipes. -- `tests/`: End-to-end and unit tests (Requires `pytest`). -- `benchmarks/`: Performance testing and evaluation suites. - -## Host Configuration - -This project requires specific host-level tuning (SSH protection, increased `inotify` limits) to prevent networking conflicts during Docker teardowns. See `DEVELOPMENT.md` for the full setup guide. +# sparkstack Agent Guide + +This file routes work to the smallest authoritative context. Do not accumulate +incident history or duplicate skill procedures here. + +## Start Here + +- Read every file in `.agents/rules/`; those rules are always in force. +- Search `.agents/skills/`, `skills/`, and + `sparkrun/sparkrun-cc-plugin/skills/` for the task's domain, then read the + matching skill completely before acting. +- Read `ARCHITECTURE.md` for component ownership and source-dependency + boundaries. Use `docs/README.md` to find durable guidance. +- Use a dedicated linked worktree for repository changes. Do not mutate + sibling source dependencies unless the task explicitly requires it and the + matching skill permits it. + +## Task Routes + +| Task | Read first | Proof | +| ----------------------------------- | -------------------------------------------------------- | ------------------------------------------------ | +| Build, rotate, or rebalance a stack | `stack-manager` | its approved plan and full verification protocol | +| Update OpenClaw or SparkRun source | `stack-upkeep`, `source-dependency-dev` | skill-defined integration proof | +| Debug a stack | `stack-debugging`, then the matching domain skill | focused diagnosis before mutation | +| Change Docker/network topology | `stack-knowledge` and Docker guidance | topology checks and incident record | +| Change observability | `monitoring` | metrics/config validation | +| Change Python orchestration | `python-pro`, `async-python-patterns`, `ARCHITECTURE.md` | focused unit/regression tests | +| Verify a running stack | `stack-verification` | ordered `tests/e2e/` evidence | +| Change source dependency workflow | `source-dependency-dev` | branch and integration evidence | + +## Ownership Boundaries + +- `sparkstack/` owns the CLI, orchestration, schemas, IPC, and health logic. +- `services/` owns Compose fragments and service-specific managers. +- `tests/unit/` and `tests/regression/` prove source behavior without a live + stack; `tests/e2e/` proves the explicitly prepared live stack. +- `sparkstack-registry` owns recipes and model configurations; its branching + rule is exceptional and defined in `.agents/rules/`. +- `sparkrun` and `openclaw` are sibling source dependencies with their own + branch and mutation policies. +- Runtime state, secrets, logs, model weights, and generated benchmark data do + not belong in this repository. + +When a durable rule changes, update the owning skill, architecture document, +or executable check. Keep this router compact. diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md new file mode 100644 index 00000000..8cd78d34 --- /dev/null +++ b/ARCHITECTURE.md @@ -0,0 +1,77 @@ +# Architecture + +sparkstack is the control plane for a local AI service stack. It turns reviewed +source and registry recipes into coordinated Docker services, emits structured +progress, and verifies the resulting system. This document identifies the +owner of each concern; it is not an operational runbook. + +## Control Flow + +```text +sparkstack CLI + |-- core schemas, discovery, environment, git, IPC + |-- manager build/update/launch/sync operations + | |-- read sparkstack-registry recipes + | |-- integrate sibling OpenClaw and SparkRun sources + | `-- render/apply services/* Compose configuration + |-- JSON-Lines stdout for automation + `-- UDS events for sparkstack status TUI + | + v + Docker service stack + OpenClaw -> LiteLLM -> vLLM backends + | + v + Prometheus/Grafana/Tempo +``` + +The CLI and TUI consume the same event model. Automated callers use `--json`; +they must not scrape Rich or Textual output. + +## Ownership Map + +| Concern | Source of truth | Evidence | +| --- | --- | --- | +| CLI surface | `sparkstack/cli/` | unit tests and command help | +| Shared configuration and events | `sparkstack/core/` | unit/regression tests | +| Lifecycle orchestration | `sparkstack/manager/` | manager unit tests and IPC regressions | +| Service definitions | `services/` | Compose/config checks and prepared-stack E2E | +| Model/stack recipes | sibling `sparkstack-registry` | registry validation and memory-law checks | +| SparkRun behavior | sibling `sparkrun` | its repository workflow plus integration tests | +| OpenClaw behavior | sibling `openclaw` | its repository workflow plus gateway E2E | +| Network topology and incident learning | `stack-knowledge` skill | live topology diagnostics | +| Deployment workflow | `stack-manager` skill | approved plan, benchmark, and E2E evidence | + +## Source and Runtime Boundaries + +- Repository source is declarative intent and orchestration code. Running + containers, `~/.openclaw`, secrets, sockets, logs, and generated stack state + are external mutable runtime state. +- The OpenClaw source checkout is distinct from `~/.openclaw` runtime state. +- Docker-out-of-Docker consumers require host-valid absolute paths and + identical volume mappings where both host and container access are needed. +- Service-to-service traffic uses shared-network names, never fixed container + IPs or host hairpin routing. +- The registry's main-only policy and sibling branch prerequisites live in + `.agents/rules/`; do not infer or duplicate them here. + +## Change Classes and Proof + +Documentation and source-only contract changes can use unit, regression, +format, lint, type, and pre-commit evidence. A change to infrastructure, +configuration, a model, or the running stack additionally requires the +ordered live E2E protocol after its prerequisites are satisfied. + +Do not run live E2E against an arbitrary environment: first satisfy the source +dependency branch rules and the selected skill's preparation steps. + +## Proof Ladder + +1. Run the narrow unit or regression test for the changed owner. +2. Run formatting, linting, type, and security checks defined by hooks. +3. Run the complete non-live unit/regression suite. +4. For infrastructure or live-stack changes, prepare the stack and run the + ordered E2E verification skill. +5. Require CI on the exact reviewed commit. +6. For deployment work, record live configuration and benchmark evidence as + required by the stack-manager plan. diff --git a/README.md b/README.md index 3bdafd9c..e6385c15 100644 --- a/README.md +++ b/README.md @@ -2,6 +2,9 @@ This repository serves as the primary deployment orchestrator for the Spark ecosystem, managing the `openclaw` backend, `sparkrun` orchestrator, and various local LLM (vLLM) backend stacks. +See [ARCHITECTURE.md](ARCHITECTURE.md) for component ownership and source/runtime +boundaries, and [docs/README.md](docs/README.md) for the documentation index. + ## Architecture The Spark Services Orchestrator (`sparkstack`) acts as the command center for the entire Spark AI ecosystem. It provides a robust, async-first Python CLI to manage the lifecycle of various interconnected services. diff --git a/docs/README.md b/docs/README.md new file mode 100644 index 00000000..4a20c3e5 --- /dev/null +++ b/docs/README.md @@ -0,0 +1,20 @@ +# Documentation Index + +Use the smallest owner for the task. + +| Need | Document | +| --- | --- | +| Understand control flow and ownership | [`../ARCHITECTURE.md`](../ARCHITECTURE.md) | +| Set up development and host tuning | [`../DEVELOPMENT.md`](../DEVELOPMENT.md) | +| See the product overview and CLI entry points | [`../README.md`](../README.md) | +| Follow mandatory repository rules | [`../.agents/rules/`](../.agents/rules/) | +| Build or rotate model stacks | [`../.agents/skills/stack-manager/SKILL.md`](../.agents/skills/stack-manager/SKILL.md) | +| Debug topology and consult incident knowledge | [`../.agents/skills/stack-knowledge/SKILL.md`](../.agents/skills/stack-knowledge/SKILL.md) | +| Verify a prepared running stack | [`../.agents/skills/stack-verification/SKILL.md`](../.agents/skills/stack-verification/SKILL.md) | +| Change sibling source dependencies | [`../.agents/skills/source-dependency-dev/SKILL.md`](../.agents/skills/source-dependency-dev/SKILL.md) | +| Change monitoring | [`../.agents/skills/monitoring/SKILL.md`](../.agents/skills/monitoring/SKILL.md) | + +Implementation lives in `sparkstack/`; service definitions live in +`services/`; executable evidence lives in `tests/`. Domain skills are the +authoritative procedures. Link new durable documentation here rather than +expanding `AGENTS.md`. diff --git a/tests/unit/test_agent_context.py b/tests/unit/test_agent_context.py new file mode 100644 index 00000000..4e6b88e2 --- /dev/null +++ b/tests/unit/test_agent_context.py @@ -0,0 +1,62 @@ +import re +from pathlib import Path + +ROOT = Path(__file__).parents[2] +MAX_CONTEXT_BYTES = 14 * 1024 + + +def read(path: str) -> str: + return (ROOT / path).read_text() + + +def test_always_loaded_agent_context_stays_compact() -> None: + assert len((ROOT / "AGENTS.md").read_bytes()) <= MAX_CONTEXT_BYTES + + +def test_agent_router_points_to_rules_skills_architecture_and_proof() -> None: + agents = read("AGENTS.md") + for route in ( + ".agents/rules/", + ".agents/skills/", + "ARCHITECTURE.md", + "docs/README.md", + "tests/e2e/", + "source-dependency-dev", + ): + assert route.lower() in agents.lower() + + +def test_router_does_not_duplicate_operational_commands() -> None: + agents = read("AGENTS.md") + assert not re.search(r"docker (compose|rm)|sparkstack (update|set-current)|git (pull|rebase)", agents) + + +def test_documentation_index_targets_exist() -> None: + for path in ( + "ARCHITECTURE.md", + "README.md", + "DEVELOPMENT.md", + ".agents/skills/stack-manager/SKILL.md", + ".agents/skills/stack-knowledge/SKILL.md", + ".agents/skills/stack-verification/SKILL.md", + ".agents/skills/source-dependency-dev/SKILL.md", + ".agents/skills/monitoring/SKILL.md", + ): + assert (ROOT / path).is_file(), path + + +def test_architecture_records_source_runtime_and_dependency_boundaries() -> None: + architecture = read("ARCHITECTURE.md") + for concept in ( + "Source and Runtime Boundaries", + "sparkstack-registry", + "sparkrun", + "openclaw", + "Proof Ladder", + ): + assert concept.lower() in architecture.lower() + + +def test_negative_fixtures_violate_contract() -> None: + assert len(b"x" * (MAX_CONTEXT_BYTES + 1)) > MAX_CONTEXT_BYTES + assert re.search(r"docker (compose|rm)", "docker compose down") From 0b354be94494566f2e4cd185385c4a5c47882eba Mon Sep 17 00:00:00 2001 From: Joe LaPenna Date: Fri, 2 Oct 2026 04:44:16 +0000 Subject: [PATCH 2/2] style: format agent context contract Refs #70 --- tests/unit/test_agent_context.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/unit/test_agent_context.py b/tests/unit/test_agent_context.py index 4e6b88e2..3c338cf4 100644 --- a/tests/unit/test_agent_context.py +++ b/tests/unit/test_agent_context.py @@ -28,7 +28,9 @@ def test_agent_router_points_to_rules_skills_architecture_and_proof() -> None: def test_router_does_not_duplicate_operational_commands() -> None: agents = read("AGENTS.md") - assert not re.search(r"docker (compose|rm)|sparkstack (update|set-current)|git (pull|rebase)", agents) + assert not re.search( + r"docker (compose|rm)|sparkstack (update|set-current)|git (pull|rebase)", agents + ) def test_documentation_index_targets_exist() -> None: