From d2c4d1f6307afd5948cd302a1928306d859daa06 Mon Sep 17 00:00:00 2001 From: dalongbao <75789627+dalongbao@users.noreply.github.com> Date: Fri, 14 Aug 2026 12:47:18 +0800 Subject: [PATCH 1/3] feat: agl-skill (#534) Co-authored-by: dalongbao --- .github/workflows/skills.yml | 33 + skills/README.md | 139 ++ .../.claude-plugin/plugin.json | 11 + skills/agent-lightning/SKILL.md | 90 + ...lightning-alfworld-success-finale-cost.svg | 1657 +++++++++++++++ ...ightning-alfworld-success-overall-cost.svg | 1675 +++++++++++++++ ...tning-officeqa-correctness-finale-cost.svg | 1790 +++++++++++++++++ ...ning-officeqa-correctness-overall-cost.svg | 1591 +++++++++++++++ ...-spreadsheetbench-accuracy-finale-cost.svg | 1659 +++++++++++++++ ...spreadsheetbench-accuracy-overall-cost.svg | 1735 ++++++++++++++++ 10 files changed, 10380 insertions(+) create mode 100644 .github/workflows/skills.yml create mode 100644 skills/README.md create mode 100644 skills/agent-lightning/.claude-plugin/plugin.json create mode 100644 skills/agent-lightning/SKILL.md create mode 100644 skills/assets/agent-lightning-alfworld-success-finale-cost.svg create mode 100644 skills/assets/agent-lightning-alfworld-success-overall-cost.svg create mode 100644 skills/assets/agent-lightning-officeqa-correctness-finale-cost.svg create mode 100644 skills/assets/agent-lightning-officeqa-correctness-overall-cost.svg create mode 100644 skills/assets/agent-lightning-spreadsheetbench-accuracy-finale-cost.svg create mode 100644 skills/assets/agent-lightning-spreadsheetbench-accuracy-overall-cost.svg diff --git a/.github/workflows/skills.yml b/.github/workflows/skills.yml new file mode 100644 index 000000000..03a143a5d --- /dev/null +++ b/.github/workflows/skills.yml @@ -0,0 +1,33 @@ +name: Validate Agent Lightning Skill + +permissions: + contents: read + +on: + push: + branches: [main] + paths: + - 'skills/**' + - '.github/workflows/skills.yml' + pull_request: + branches: [main] + paths: + - 'skills/**' + - '.github/workflows/skills.yml' + workflow_dispatch: + +jobs: + validate: + name: Validate Agent Skills and Claude plugin formats + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - uses: actions/checkout@v6 + - uses: astral-sh/setup-uv@v7 + - name: Validate Agent Skills format + run: uvx --from 'skills-ref==0.1.1' agentskills validate skills/agent-lightning + - uses: actions/setup-node@v6 + with: + node-version: '22' + - name: Validate Claude Code plugin + run: npx --yes @anthropic-ai/claude-code@2.1.218 plugin validate skills/agent-lightning diff --git a/skills/README.md b/skills/README.md new file mode 100644 index 000000000..f831ffe07 --- /dev/null +++ b/skills/README.md @@ -0,0 +1,139 @@ +# Agent Skills + +Skills in the [Agent Skills](https://agentskills.io) format (`/SKILL.md`), installable into any compatible agent. + +## Agent Lightning + +Turns your coding agent into an **agent optimizer**: given an editable agent and a benchmark to hillclimb on, it improves the agent's accuracy, cost, and latency through focused, individually-measured edits — keeping only what moves the frontier. It was measured against a no-skill control under a fair, leakage-free protocol. + +You provide the environment; the skill does the optimizing. Before invoking it, have ready: a working copy of the agent (keep the original pristine), labeled examples, a frozen eval command, and an objective + budget. + +### Installation + +Install the skill from this repository for Claude Code, Codex, or GitHub Copilot: + +```bash +gh skill install microsoft/agent-lightning agent-lightning --agent claude-code +gh skill install microsoft/agent-lightning agent-lightning --agent codex +gh skill install microsoft/agent-lightning agent-lightning --agent github-copilot +``` + +The `skills/agent-lightning/` directory is both the canonical Agent Skills package and the Claude Code plugin root, so both publication paths use the same `SKILL.md` without a copied or symlinked wrapper. + +### Results + +**Main finding:** Coding-agent harnesses are already strong optimizers. The clearest opportunity is improving consistency while preserving their high average performance, rather than expecting large score gains. + +SkillOpt and the other non-agentic results are taken from the [SkillOpt paper](https://github.com/microsoft/SkillOpt) (Table 1); our agentic rows use the same splits and average all optimizers, budgets, and replicates. + +| Method | SpreadsheetBench accuracy (%) | OfficeQA correctness (%) | ALFWorld success (%) | +| :--- | ---: | ---: | ---: | +| No skill | 36.1 | 22.1 | 73.1 | +| Human skill | 42.9 | 45.9 | 56.7 | +| LLM skill | 36.8 | 36.6 | 65.7 | +| Trace2Skill | 40.7 | 20.9 | 82.8 | +| TextGrad | 38.2 | 30.0 | 70.9 | +| GEPA | 42.5 | 45.3 | 81.3 | +| SkillOpt | 47.5 | 48.8 | 85.8 | +| Agentic optimizer average, no skill | 62.9 | 54.1 | 88.6 | +| **Agentic optimizer average, Agent Lightning** | **66.7** | **54.5** | **94.9** | + +#### Performance versus overall cost + +Each benchmark includes the \$5, \$10, and \$25 nominal-budget groups with three runs per treatment cell. Every point is one held-out finale result: the x-axis is that run's overall cost on a log scale, and the y-axis is SpreadsheetBench accuracy, OfficeQA correctness, or ALFWorld success. Color and shape identify the optimizer; filled markers use Agent Lightning and hollow markers are no-skill controls. Budget is not encoded in the legend. Overall cost includes optimizer LLM calls, train/self-evaluation, and held-out finale deployment; it excludes the pristine-baseline evaluations. + +Claude Code uses Claude Opus 4.8; Codex and GitHub Copilot use GPT 5.6 Sol as their optimizer models. + +![SpreadsheetBench accuracy versus overall cost](assets/agent-lightning-spreadsheetbench-accuracy-overall-cost.svg) + +![OfficeQA correctness versus overall cost](assets/agent-lightning-officeqa-correctness-overall-cost.svg) + +![ALFWorld success versus overall cost](assets/agent-lightning-alfworld-success-overall-cost.svg) + +#### Performance versus finale cost + +The selected-budget views use the groups with the strongest aggregate skill-over-control lift: \$5 for SpreadsheetBench and \$10 for OfficeQA and ALFWorld. Every harness/treatment point is one of three runs; the x-axis is that run's finale cost, and the y-axis is held-out SpreadsheetBench accuracy, OfficeQA correctness, or ALFWorld success. Finale cost measures LLM gateway spend, so an ALFWorld deterministic controller can have exactly \$0 finale cost while still executing and scoring real environment steps; coincident zero-cost ALFWorld results are offset slightly along the x-axis so each replicate remains visible. SpreadsheetBench and OfficeQA show their aggregate pristine-baseline results as single reference points. The corrected ALFWorld records do not include baseline deployment cost, so its aggregate measured success is shown as a horizontal reference instead of assigning it an x-coordinate. + +![SpreadsheetBench accuracy versus finale cost](assets/agent-lightning-spreadsheetbench-accuracy-finale-cost.svg) + +![OfficeQA correctness versus finale cost](assets/agent-lightning-officeqa-correctness-finale-cost.svg) + +![ALFWorld success versus finale cost](assets/agent-lightning-alfworld-success-finale-cost.svg) + +#### \$5 budget snapshot + +| Benchmark metric (train/test) | Result | Score (%) | Finale cost | +| :--- | :--- | ---: | ---: | +| SpreadsheetBench accuracy (120/280) | Baseline | 25.66 ± 2.65 | \$1.51 ± 0.04 | +| | Claude Code with skill | 63.79 ± 5.24 | **\$2.45 ± 0.45** | +| | Claude Code without skill | **68.23 ± 0.55** | \$2.52 ± 0.93 | +| | Codex with skill | **65.47 ± 4.59** | **\$1.66 ± 0.19** | +| | Codex without skill | 41.49 ± 24.28 | \$1.73 ± 0.15 | +| | Copilot with skill | **66.31 ± 2.05** | **\$1.65 ± 0.13** | +| | Copilot without skill | 51.68 ± 20.82 | \$1.66 ± 0.09 | +| OfficeQA correctness (50/172) | Baseline | 31.78 ± 1.21 | \$2.78 ± 0.06 | +| | Claude Code with skill | 56.78 ± 3.87 | \$5.35 ± 1.60 | +| | Claude Code without skill | **59.69 ± 4.88** | **\$4.69 ± 1.60** | +| | Codex with skill | **49.81 ± 2.98** | **\$3.38 ± 0.54** | +| | Codex without skill | 49.61 ± 0.67 | \$3.77 ± 0.22 | +| | Copilot with skill | 51.55 ± 3.74 | **\$3.69 ± 0.45** | +| | Copilot without skill | **54.65 ± 2.01** | \$4.20 ± 0.58 | +| ALFWorld success (3553/134) | Baseline | 56.97 ± 0.43 | — | +| | Claude Code with skill | **95.02 ± 1.14** | \$3.83 ± 0.78 | +| | Claude Code without skill | 93.53 ± 4.11 | **\$3.21 ± 0.38** | +| | Codex with skill | 87.31 ± 21.97 | \$1.69 ± 2.92 | +| | Codex without skill | **96.52 ± 3.02** | **\$0.97 ± 1.68** | +| | Copilot with skill | **99.75 ± 0.43** | \$0.01 ± 0.02 | +| | Copilot without skill | 95.52 ± 7.12 | **\$0.00 ± 0.00** | + +#### \$10 budget snapshot + +| Benchmark metric (train/test) | Result | Score (%) | Finale cost | +| :--- | :--- | ---: | ---: | +| SpreadsheetBench accuracy (120/280) | Baseline | 25.66 ± 2.65 | \$1.51 ± 0.04 | +| | Claude Code with skill | 67.75 ± 2.40 | **\$2.01 ± 0.26** | +| | Claude Code without skill | **69.42 ± 4.32** | \$2.11 ± 0.21 | +| | Codex with skill | 64.63 ± 0.75 | **\$1.59 ± 0.06** | +| | Codex without skill | **68.59 ± 1.16** | \$1.72 ± 0.08 | +| | Copilot with skill | **69.30 ± 4.32** | \$2.08 ± 0.74 | +| | Copilot without skill | 64.39 ± 2.88 | **\$1.63 ± 0.05** | +| OfficeQA correctness (50/172) | Baseline | 31.78 ± 1.21 | \$2.78 ± 0.06 | +| | Claude Code with skill | **62.60 ± 3.74** | \$5.88 ± 0.82 | +| | Claude Code without skill | 59.30 ± 1.74 | **\$5.68 ± 0.74** | +| | Codex with skill | **54.07 ± 1.16** | \$3.95 ± 0.43 | +| | Codex without skill | 50.00 ± 0.58 | **\$3.77 ± 0.27** | +| | Copilot with skill | **53.68 ± 0.89** | \$3.46 ± 0.03 | +| | Copilot without skill | 51.16 ± 4.07 | **\$3.07 ± 1.46** | +| ALFWorld success (3553/134) | Baseline | 56.97 ± 0.43 | — | +| | Claude Code with skill | 93.78 ± 0.43 | **\$3.62 ± 0.51** | +| | Claude Code without skill | **94.28 ± 3.02** | \$3.67 ± 0.90 | +| | Codex with skill | **99.00 ± 0.86** | \$0.75 ± 1.28 | +| | Codex without skill | 89.55 ± 18.10 | **\$0.00 ± 0.00** | +| | Copilot with skill | **100.00 ± 0.00** | \$0.00 ± 0.00 | +| | Copilot without skill | 66.92 ± 57.30 | \$0.00 ± 0.00 | + +#### \$25 budget snapshot + +| Benchmark metric (train/test) | Result | Score (%) | Finale cost | +| :--- | :--- | ---: | ---: | +| SpreadsheetBench accuracy (120/280) | Baseline | 25.66 ± 2.65 | \$1.51 ± 0.04 | +| | Claude Code with skill | **71.70 ± 7.11** | \$8.12 ± 5.26 | +| | Claude Code without skill | 68.94 ± 1.98 | **\$5.31 ± 5.71** | +| | Codex with skill | 62.95 ± 2.52 | **\$1.76 ± 0.24** | +| | Codex without skill | **65.23 ± 2.40** | \$1.84 ± 0.24 | +| | Copilot with skill | **68.71 ± 3.12** | \$3.48 ± 3.17 | +| | Copilot without skill | 68.47 ± 3.60 | **\$1.69 ± 0.01** | +| OfficeQA correctness (50/172) | Baseline | 31.78 ± 1.21 | \$2.78 ± 0.06 | +| | Claude Code with skill | **60.27 ± 4.70** | **\$5.32 ± 0.62** | +| | Claude Code without skill | 57.17 ± 2.98 | \$11.81 ± 7.02 | +| | Codex with skill | 50.78 ± 2.87 | **\$3.28 ± 0.22** | +| | Codex without skill | **52.13 ± 0.34** | \$3.83 ± 0.41 | +| | Copilot with skill | 51.16 ± 1.74 | **\$4.13 ± 0.25** | +| | Copilot without skill | **53.10 ± 0.34** | \$4.38 ± 0.40 | +| ALFWorld success (3553/134) | Baseline | 56.97 ± 0.43 | — | +| | Claude Code with skill | 82.59 ± 21.84 | \$4.28 ± 1.27 | +| | Claude Code without skill | **94.78 ± 4.48** | **\$3.02 ± 0.55** | +| | Codex with skill | **96.77 ± 5.60** | \$1.08 ± 1.87 | +| | Codex without skill | 66.67 ± 57.74 | **\$0.01 ± 0.02** | +| | Copilot with skill | 100.00 ± 0.00 | \$0.12 ± 0.20 | +| | Copilot without skill | 100.00 ± 0.00 | **\$0.00 ± 0.00** | diff --git a/skills/agent-lightning/.claude-plugin/plugin.json b/skills/agent-lightning/.claude-plugin/plugin.json new file mode 100644 index 000000000..a7e076663 --- /dev/null +++ b/skills/agent-lightning/.claude-plugin/plugin.json @@ -0,0 +1,11 @@ +{ + "name": "agent-lightning", + "displayName": "Agent Lightning", + "description": "Turns coding agents into agent optimizers that improve accuracy, cost, latency, and reliability against a benchmark.", + "author": { + "name": "Agent Lightning Team" + }, + "repository": "https://github.com/microsoft/agent-lightning", + "homepage": "https://microsoft.github.io/agent-lightning/", + "license": "MIT" +} diff --git a/skills/agent-lightning/SKILL.md b/skills/agent-lightning/SKILL.md new file mode 100644 index 000000000..128cc2155 --- /dev/null +++ b/skills/agent-lightning/SKILL.md @@ -0,0 +1,90 @@ +--- +name: agent-lightning +description: >- + Provides the action space, tradeoffs, and evaluation context for improving an + editable AI agent against a benchmark while preserving its deployment contract. + Use when optimizing agent accuracy, cost, latency, or reliability. +--- + +# Agent Lightning + +Agent optimization is a search over interacting choices. The useful question is +not which architecture is most sophisticated, but which change moves the requested +accuracy, cost, latency, and reliability frontier for this agent. + +The optimizer's development budget and the resulting agent's per-run cost are +different quantities. More development budget creates room to learn; it does not +imply that the deployed agent should spend more on every task. + +Your remaining development budget is reported at `/artifacts/cost_budget.json` +(`{spent, budget, remaining}`), refreshed as you work. Read that file to see how +much is left, and keep iterating — measure, edit, re-score — while meaningful +budget remains; do not stop at the first plausible result. A run is finished not +because one change worked, but because further measured changes no longer improve +the frontier within the budget you still have. If `remaining` is large, there is +more search to do: ground more cases, probe a lever you have not tested, or add +reps to resolve a noisy comparison. Check `remaining` again after each expensive +step so the decision to stop is evidence-based, not a default. + +## Action space + +| Lever | What it changes | Useful signal | Main tradeoff | +| ------------------------ | -------------------------------------------------------- | --------------------------------------------------------------- | ------------------------------------------ | +| Input grounding | Information and state visible to the model | Relevant deployment-visible context is missing | Longer context can distract or cost more | +| Output contract | Representation, types, schema, files, and terminal state | Work looks reasonable but is rejected or unreadable | Can overfit evaluator quirks | +| Prompt | Interpretation, priorities, and constraints | Instructions are misunderstood or important details are ignored | Prompt gains can be brittle | +| Tools | Deterministic inspection, computation, and execution | The model is approximating work a tool can do reliably | More code and new failure modes | +| Model | Base capability and knowledge | The primary cannot solve grounded cases | Cost, latency, and availability | +| Reasoning effort | Computation used by the primary call | Grounded hard cases remain | Cost and latency can grow nonlinearly | +| Failure isolation | Whether one failure damages other work | Individual tasks crash, time out, or corrupt shared state | Isolation does not recover the failed task | +| Conditional repair | A second attempt informed by failure evidence | Deployment-visible checks expose a recoverable failure | Extra calls and possible regressions | +| Routing | Different handling for different task classes | Difficulty or failure risk varies predictably | Router mistakes and operational complexity | +| Planning and interaction | State, ordering, and tool use across multiple steps | Long tasks lose goals or ignore observations | More state and control-flow overhead | +| Retrieval | Facts supplied from an available corpus | Correct answers depend on external knowledge | Retrieval errors and added latency | +| Critique or selection | Additional views or candidates | Independent attempts expose different useful information | Multiplied calls, cost, and latency | + +## Reading the evidence + +Different failures expose different amounts of information: + +- Exceptions, missing artifacts, invalid schemas, and timeouts are objective + signals. They can support deterministic checks or focused recovery. +- A valid-looking but wrong answer may expose no label-free repair signal. More + calls with the same information can repeat the same mistake. +- Repeated failures across different attempts suggest shared blindness, a contract + mismatch, or missing capability. Diverse failures make routing, critique, or + selection more plausible. +- A development gain that disappears under validation may come from randomness, + memorized examples, training-only fields, or a different deployment path. +- An unavailable measurement is unknown evidence, not proof that a candidate + improved or regressed. + +Levers interact. More reasoning cannot recover information the model never sees. +Repair cannot fix a systematic contract error when the retry receives no new +evidence. Failure isolation preserves the batch but does not repair an item. A +global model or effort increase and conditional escalation occupy different points +on the frontier. + +## Evaluation context + +Agent evaluations are often stochastic. A score can improve while many individual +cases regress, and a single strong result can be a lucky draw. Fixed cases, +validation splits, repeated runs, frozen primary outputs, and small probes are +different ways to reduce uncertainty; their value depends on the decision and the +available budget. + +Comparisons are easiest to interpret when the checkpoint, cases, model, effort, +concurrency, scorer, and execution path are held constant except for the variable +being studied. Accuracy is only one result: completion rate, downside, cost, latency, +and variance can change the decision. + +Development may expose labels, metadata, or tools that do not exist at deployment. +An improvement that depends on them is not a deployed improvement. Likewise, +measurement plumbing can fail independently of the agent being tested. + +## Boundaries + +- Preserve the target's external interface and deployment environment. +- Do not expose held-out labels or training-only answer fields to the deployed path. +- Treat the scorer and evaluation contract as immutable measurement surfaces. +- Leave a coherent measured checkpoint, not an unfinished or partially tested edit. diff --git a/skills/assets/agent-lightning-alfworld-success-finale-cost.svg b/skills/assets/agent-lightning-alfworld-success-finale-cost.svg new file mode 100644 index 000000000..ae8f1efde --- /dev/null +++ b/skills/assets/agent-lightning-alfworld-success-finale-cost.svg @@ -0,0 +1,1657 @@ + + + + + + + + 2026-08-11T20:51:34.507463 + image/svg+xml + + + Matplotlib v3.11.0, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/skills/assets/agent-lightning-alfworld-success-overall-cost.svg b/skills/assets/agent-lightning-alfworld-success-overall-cost.svg new file mode 100644 index 000000000..06b1a6242 --- /dev/null +++ b/skills/assets/agent-lightning-alfworld-success-overall-cost.svg @@ -0,0 +1,1675 @@ + + + + + + + + 2026-08-11T18:08:59.923766 + image/svg+xml + + + Matplotlib v3.11.1, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/skills/assets/agent-lightning-officeqa-correctness-finale-cost.svg b/skills/assets/agent-lightning-officeqa-correctness-finale-cost.svg new file mode 100644 index 000000000..e248fa869 --- /dev/null +++ b/skills/assets/agent-lightning-officeqa-correctness-finale-cost.svg @@ -0,0 +1,1790 @@ + + + + + + + + 2026-08-11T20:51:34.456714 + image/svg+xml + + + Matplotlib v3.11.0, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/skills/assets/agent-lightning-officeqa-correctness-overall-cost.svg b/skills/assets/agent-lightning-officeqa-correctness-overall-cost.svg new file mode 100644 index 000000000..acc80f826 --- /dev/null +++ b/skills/assets/agent-lightning-officeqa-correctness-overall-cost.svg @@ -0,0 +1,1591 @@ + + + + + + + + 2026-08-11T18:08:59.868856 + image/svg+xml + + + Matplotlib v3.11.1, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/skills/assets/agent-lightning-spreadsheetbench-accuracy-finale-cost.svg b/skills/assets/agent-lightning-spreadsheetbench-accuracy-finale-cost.svg new file mode 100644 index 000000000..002fb5ae4 --- /dev/null +++ b/skills/assets/agent-lightning-spreadsheetbench-accuracy-finale-cost.svg @@ -0,0 +1,1659 @@ + + + + + + + + 2026-08-11T20:51:34.373170 + image/svg+xml + + + Matplotlib v3.11.0, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/skills/assets/agent-lightning-spreadsheetbench-accuracy-overall-cost.svg b/skills/assets/agent-lightning-spreadsheetbench-accuracy-overall-cost.svg new file mode 100644 index 000000000..d5cb27dd3 --- /dev/null +++ b/skills/assets/agent-lightning-spreadsheetbench-accuracy-overall-cost.svg @@ -0,0 +1,1735 @@ + + + + + + + + 2026-08-11T18:08:59.753083 + image/svg+xml + + + Matplotlib v3.11.1, https://matplotlib.org/ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + From ba07ca0b3d5397b6e4fbb73a3432286937fe3f50 Mon Sep 17 00:00:00 2001 From: REISEN2330 <181534190+REISEN2330@users.noreply.github.com> Date: Sun, 16 Aug 2026 21:12:51 +0800 Subject: [PATCH 2/3] feat(examples): add LangGraph tracing example with AgentOps Fixes #535 --- .github/workflows/tests.yml | 18 +++++ examples/langgraph/README.md | 27 +++++++ examples/langgraph/trace_langgraph.py | 105 ++++++++++++++++++++++++++ 3 files changed, 150 insertions(+) create mode 100644 examples/langgraph/README.md create mode 100644 examples/langgraph/trace_langgraph.py diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index fbe077451..73469d980 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -236,3 +236,21 @@ jobs: run: cd dashboard && npm ci - name: Run vitest run: cd dashboard && npm run vitest + + langgraph-example: + name: LangGraph Example + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@v6 + - uses: astral-sh/setup-uv@v7 + with: + enable-cache: true + python-version: '3.12' + - name: Sync dependencies + run: uv sync --frozen --no-default-groups --group langchain + - name: Run LangGraph tracing example + run: | + source .venv/bin/activate + cd examples/langgraph + python trace_langgraph.py diff --git a/examples/langgraph/README.md b/examples/langgraph/README.md new file mode 100644 index 000000000..8b8720a4b --- /dev/null +++ b/examples/langgraph/README.md @@ -0,0 +1,27 @@ +# LangGraph Tracing Example + +This example traces a tiny [LangGraph](https://github.com/langchain-ai/langgraph) +workflow with Agent Lightning's AgentOps tracer and verifies that every +graph node is captured as a span in the store — the first step toward +training agents built with LangChain/LangGraph. + +The run is fully offline: the chat model is LangChain's deterministic +`FakeMessagesListChatModel`, so no API keys, GPU, or network access are +required. + +## Run + +```bash +uv sync --frozen --group dev --group langchain --no-default-groups +python examples/langgraph/trace_langgraph.py +``` + +Expected output: the final graph state plus the captured span list. The +script exits `0` after asserting that the LangGraph workflow execution +and at least one model call were captured in the store. + +## Included Files + +- `trace_langgraph.py` — builds the two-node graph, runs it under the + AgentOps tracer inside a rollout, then reads the spans back from the + in-memory store and asserts the workflow and model call were captured. diff --git a/examples/langgraph/trace_langgraph.py b/examples/langgraph/trace_langgraph.py new file mode 100644 index 000000000..d4d6c35b0 --- /dev/null +++ b/examples/langgraph/trace_langgraph.py @@ -0,0 +1,105 @@ +# Copyright (c) Microsoft. All rights reserved. + +"""Trace a LangGraph agent with Agent Lightning. + +This example runs a tiny two-node [LangGraph](https://github.com/langchain-ai/langgraph) +workflow — one deterministic chat-model node and one plain Python node — +under Agent Lightning's AgentOps tracer, then reads the captured spans +back from the store. + +The whole run is offline: the chat model is LangChain's deterministic +``FakeMessagesListChatModel``, so no API keys, GPU, or network calls are +required. + +Run it with: + +```bash +uv sync --frozen --group dev --group langchain --no-default-groups +python examples/langgraph/trace_langgraph.py +``` +""" + +import asyncio +from typing import Any, Dict + +try: + # langchain-core >= 1.0 moved the fake chat models here. + from langchain_core.fakes import FakeMessagesListChatModel +except ImportError: # pragma: no cover - older langchain-core + from langchain_core.language_models.fake_chat_models import FakeMessagesListChatModel + +from langchain_core.messages import AIMessage, HumanMessage +from langgraph.graph import END, START, MessagesState, StateGraph +from langgraph.graph.state import CompiledStateGraph +from rich.console import Console + +from agentlightning import AgentOpsTracer, setup_logging +from agentlightning.store import InMemoryLightningStore + +console = Console() + + +def build_graph() -> CompiledStateGraph: + """Build a two-node LangGraph workflow. + + ``say_hello`` runs a deterministic fake chat model; ``reverse_text`` + is a plain Python node. After the run, the workflow execution and the + model call must both show up as captured spans. + """ + llm = FakeMessagesListChatModel(responses=[AIMessage(content="Hello from the fake model!")]) + + def say_hello(state: MessagesState) -> Dict[str, Any]: + """Ask the (fake) chat model for a greeting.""" + response = llm.invoke(state["messages"]) + return {"messages": [response]} + + def reverse_text(state: MessagesState) -> Dict[str, Any]: + """Reverse the last message's content without any model.""" + last = state["messages"][-1] + return {"messages": [AIMessage(content=last.content[::-1])]} # type: ignore + + graph = StateGraph(MessagesState) + graph.add_node("say_hello", say_hello) + graph.add_node("reverse_text", reverse_text) + graph.add_edge(START, "say_hello") + graph.add_edge("say_hello", "reverse_text") + graph.add_edge("reverse_text", END) + return graph.compile() + + +async def main() -> None: + """Run the graph under the AgentOps tracer and verify the spans.""" + setup_logging() + + tracer = AgentOpsTracer(agentops_managed=True, instrument_managed=False) + store = InMemoryLightningStore() + rollout = await store.start_rollout(input={"origin": "langgraph_tracing_example"}) + graph = build_graph() + + with tracer.lifespan(store): + async with tracer.trace_context( + "langgraph-run", + rollout_id=rollout.rollout_id, + attempt_id=rollout.attempt.attempt_id, + ): + handler = tracer.get_langchain_handler() + result = graph.invoke( + {"messages": [HumanMessage(content="Hello!")]}, + {"callbacks": [handler]} if handler else None, + ) + console.print(result) + + spans = await store.query_spans(rollout_id=rollout.rollout_id) + console.print(spans) + + span_names = [span.name for span in spans] + # The exact span names are owned by the AgentOps instrumentation, so + # assert on the structural guarantees instead: the workflow execution + # is captured and at least one model call was traced. + assert "langgraph.workflow.execute" in span_names, span_names + assert any("llm" in name or "model" in name for name in span_names), span_names + console.print("[green]The LangGraph workflow and its model call were captured as spans.[/green]") + + +if __name__ == "__main__": + asyncio.run(main()) From e320ed13ba41e4db4c4cf1f86ed47f2ec4df6048 Mon Sep 17 00:00:00 2001 From: REISEN2330 <181534190+REISEN2330@users.noreply.github.com> Date: Sun, 16 Aug 2026 21:19:51 +0800 Subject: [PATCH 3/3] address review: type-safe content reversal, structural span asserts, align README with CI --- examples/langgraph/README.md | 2 +- examples/langgraph/trace_langgraph.py | 12 +++++++----- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/examples/langgraph/README.md b/examples/langgraph/README.md index 8b8720a4b..bf8c0b0bb 100644 --- a/examples/langgraph/README.md +++ b/examples/langgraph/README.md @@ -12,7 +12,7 @@ required. ## Run ```bash -uv sync --frozen --group dev --group langchain --no-default-groups +uv sync --frozen --no-default-groups --group langchain python examples/langgraph/trace_langgraph.py ``` diff --git a/examples/langgraph/trace_langgraph.py b/examples/langgraph/trace_langgraph.py index d4d6c35b0..cadd44f59 100644 --- a/examples/langgraph/trace_langgraph.py +++ b/examples/langgraph/trace_langgraph.py @@ -56,7 +56,8 @@ def say_hello(state: MessagesState) -> Dict[str, Any]: def reverse_text(state: MessagesState) -> Dict[str, Any]: """Reverse the last message's content without any model.""" last = state["messages"][-1] - return {"messages": [AIMessage(content=last.content[::-1])]} # type: ignore + assert isinstance(last.content, str), f"Expected text content, got {type(last.content)}" + return {"messages": [AIMessage(content=last.content[::-1])]} graph = StateGraph(MessagesState) graph.add_node("say_hello", say_hello) @@ -93,10 +94,11 @@ async def main() -> None: console.print(spans) span_names = [span.name for span in spans] - # The exact span names are owned by the AgentOps instrumentation, so - # assert on the structural guarantees instead: the workflow execution - # is captured and at least one model call was traced. - assert "langgraph.workflow.execute" in span_names, span_names + # The exact span names are owned by the AgentOps instrumentation and + # may change between versions, so assert on structural guarantees + # instead: at least one span carries the langgraph workflow marker + # and at least one model call was traced. + assert any("langgraph" in name for name in span_names), span_names assert any("llm" in name or "model" in name for name in span_names), span_names console.print("[green]The LangGraph workflow and its model call were captured as spans.[/green]")