diff --git a/.github/DEPRECATION.md b/.github/DEPRECATION.md index 6a803ac4..521950d8 100644 --- a/.github/DEPRECATION.md +++ b/.github/DEPRECATION.md @@ -50,8 +50,7 @@ Immediate removal is allowed only for security or compliance issues. The removal - `.github/skills/internal-data-registry/SKILL.md`: **Removed**. Retired from the live catalog after confirmation that no live references remained. - `.github/scripts/bootstrap-copilot-config.sh`: **Removed**. Replaced by the - `local-sync-global-copilot-configs-into-repo` agent and the - `local-agent-sync-global-copilot-configs-into-repo` skill workflow. + `local-sync-repos` agent and skill workflow. - `.github/skills/internal-terraform-feature/SKILL.md`: **Removed**. Merged into `.github/skills/internal-terraform/SKILL.md`. - `.github/skills/internal-terraform-module/SKILL.md`: **Removed**. Merged into diff --git a/.github/INVENTORY.md b/.github/INVENTORY.md index 8f1ec343..5831aae4 100644 --- a/.github/INVENTORY.md +++ b/.github/INVENTORY.md @@ -39,6 +39,10 @@ This file is the exact path inventory for the live GitHub Copilot catalog in thi - `.github/skills/agent-os-inject-standards/SKILL.md` - `.github/skills/agent-os-plan-product/SKILL.md` - `.github/skills/agent-os-shape-spec/SKILL.md` +- `.github/skills/anthropic-docx/SKILL.md` +- `.github/skills/anthropic-pdf/SKILL.md` +- `.github/skills/anthropic-pptx/SKILL.md` +- `.github/skills/anthropic-xlsx/SKILL.md` - `.github/skills/antigravity-api-design-principles/SKILL.md` - `.github/skills/antigravity-aws-cost-optimizer/SKILL.md` - `.github/skills/antigravity-cloudformation-best-practices/SKILL.md` @@ -65,12 +69,12 @@ This file is the exact path inventory for the live GitHub Copilot catalog in thi - `.github/skills/internal-aws-mcp-research/SKILL.md` - `.github/skills/internal-aws-operations/SKILL.md` - `.github/skills/internal-aws-organization-structure/SKILL.md` -- `.github/skills/internal-aws-strategic/SKILL.md` +- `.github/skills/internal-aws/SKILL.md` - `.github/skills/internal-azure-devops/SKILL.md` - `.github/skills/internal-azure-governance/SKILL.md` - `.github/skills/internal-azure-operations/SKILL.md` - `.github/skills/internal-azure-organization-structure/SKILL.md` -- `.github/skills/internal-azure-strategic/SKILL.md` +- `.github/skills/internal-azure/SKILL.md` - `.github/skills/internal-bash-script/SKILL.md` - `.github/skills/internal-bash/SKILL.md` - `.github/skills/internal-changelog-automation/SKILL.md` @@ -82,6 +86,7 @@ This file is the exact path inventory for the live GitHub Copilot catalog in thi - `.github/skills/internal-devops-core-principles/SKILL.md` - `.github/skills/internal-docker/SKILL.md` - `.github/skills/internal-excel/SKILL.md` +- `.github/skills/internal-gateway-codebase-improvement/SKILL.md` - `.github/skills/internal-gateway-critical-master/SKILL.md` - `.github/skills/internal-gateway-execute-plans/SKILL.md` - `.github/skills/internal-gateway-idea/SKILL.md` @@ -90,13 +95,13 @@ This file is the exact path inventory for the live GitHub Copilot catalog in thi - `.github/skills/internal-gcp-governance/SKILL.md` - `.github/skills/internal-gcp-operations/SKILL.md` - `.github/skills/internal-gcp-organization-structure/SKILL.md` -- `.github/skills/internal-gcp-strategic/SKILL.md` +- `.github/skills/internal-gcp/SKILL.md` - `.github/skills/internal-github-action-composite/SKILL.md` - `.github/skills/internal-github-actions/SKILL.md` - `.github/skills/internal-github-governance/SKILL.md` - `.github/skills/internal-github-operations/SKILL.md` - `.github/skills/internal-github-pr/SKILL.md` -- `.github/skills/internal-github-strategic/SKILL.md` +- `.github/skills/internal-github/SKILL.md` - `.github/skills/internal-go/SKILL.md` - `.github/skills/internal-java-project/SKILL.md` - `.github/skills/internal-java-spring-boot-development/SKILL.md` @@ -122,18 +127,26 @@ This file is the exact path inventory for the live GitHub Copilot catalog in thi - `.github/skills/internal-terraform/SKILL.md` - `.github/skills/internal-yaml/SKILL.md` - `.github/skills/local-agent-sync-external-resources/SKILL.md` -- `.github/skills/local-agent-sync-global-copilot-configs-into-repo/SKILL.md` - `.github/skills/local-agent-sync-install-ai-resources/SKILL.md` - `.github/skills/local-copilot-log-analyzer/SKILL.md` +- `.github/skills/local-sync-repos/SKILL.md` +- `.github/skills/mattpocock-code-review/SKILL.md` +- `.github/skills/mattpocock-codebase-design/SKILL.md` +- `.github/skills/mattpocock-domain-modeling/SKILL.md` +- `.github/skills/mattpocock-grill-with-docs/SKILL.md` - `.github/skills/mattpocock-handoff/SKILL.md` +- `.github/skills/mattpocock-implement/SKILL.md` +- `.github/skills/mattpocock-improve-codebase-architecture/SKILL.md` - `.github/skills/mattpocock-research/SKILL.md` -- `.github/skills/openai-docx/SKILL.md` +- `.github/skills/mattpocock-setup-matt-pocock-skills/SKILL.md` +- `.github/skills/mattpocock-tdd/SKILL.md` +- `.github/skills/mattpocock-to-spec/SKILL.md` +- `.github/skills/mattpocock-wayfinder/SKILL.md` +- `.github/skills/mattpocock-writing-great-skills/SKILL.md` +- `.github/skills/openai-docs/SKILL.md` - `.github/skills/openai-gh-address-comments/SKILL.md` - `.github/skills/openai-gh-fix-ci/SKILL.md` -- `.github/skills/openai-pdf/SKILL.md` -- `.github/skills/openai-skill-creator/SKILL.md` -- `.github/skills/openai-slides/SKILL.md` -- `.github/skills/openai-spreadsheet/SKILL.md` +- `.github/skills/search-company-knowledge/SKILL.md` - `.github/skills/superpowers-brainstorming/SKILL.md` - `.github/skills/superpowers-dispatching-parallel-agents/SKILL.md` - `.github/skills/superpowers-executing-plans/SKILL.md` @@ -151,9 +164,9 @@ This file is the exact path inventory for the live GitHub Copilot catalog in thi - `.github/skills/terraform-terraform-test/SKILL.md` - `.github/skills/vercel-find-skills/SKILL.md` -### Support-only imported office skills +### Support-only imported document skills -These imported `openai-*` office skills remain support-only depth for repositories that explicitly need document workflows. +These vendor-prefixed imported document skills remain support-only depth for repositories that explicitly need document workflows. ## Scripts @@ -174,26 +187,24 @@ These imported `openai-*` office skills remain support-only depth for repositori - `.github/scripts/lib/repo_paths.py` - `.github/scripts/lib/shared.py` - `.github/scripts/lib/sync_exclusions.py` -- `.github/scripts/lib/syncing.py` - `.github/scripts/lib/token_risks.py` - `.github/scripts/run.sh` -- `.github/scripts/sync_copilot_catalog.py` - `.github/scripts/validate_internal_skills.py` ## Agents - `.github/agents/internal-gateway-critical-master.agent.md` +- `.github/agents/internal-gateway-execute-plans.agent.md` - `.github/agents/internal-gateway-idea.agent.md` -- `.github/agents/internal-gateway-review.agent.md` +- `.github/agents/internal-gateway-review-code.agent.md` +- `.github/agents/internal-gateway-review-generic.agent.md` - `.github/agents/internal-gateway-simple-task.agent.md` -- `.github/agents/internal-review-code.agent.md` - `.github/agents/local-sync-external-resources.agent.md` -- `.github/agents/local-sync-global-copilot-configs-into-repo.agent.md` - `.github/agents/local-sync-install-ai-resources.agent.md` +- `.github/agents/local-sync-repos.agent.md` ## Prompts - `.github/prompts/internal-architecture-md-creator.prompt.md` - `.github/prompts/internal-mega-review.prompt.md` - `.github/prompts/internal-review-ai-resources.prompt.md` -- `.github/prompts/internal-sync-plan.prompt.md` diff --git a/.github/README.md b/.github/README.md index b895198b..0c6f25c3 100644 --- a/.github/README.md +++ b/.github/README.md @@ -10,11 +10,11 @@ customization assets maintained in `cloud-strategy.github`. ## Agents - Canonical repository-owned gateway agents: - `internal-gateway-idea`, `internal-gateway-review`, + `internal-gateway-idea`, `internal-gateway-review-generic`, `internal-gateway-critical-master`, `internal-gateway-simple-task` - Specialist repository-owned review agents: - `internal-review-code` for code-focused review before merge or follow-up action. -- Approved `extended` retained plans execute through + `internal-gateway-review-code` for code-focused review before merge or follow-up action. +- Approved retained plans under tmp/superpowers/plans/ execute through `internal-gateway-execute-plans`. - When extra provenance helps, offer it as an optional follow-up detail and accept number-only replies. diff --git a/.github/agents/README.md b/.github/agents/README.md index b8aff5f6..81ad20ba 100644 --- a/.github/agents/README.md +++ b/.github/agents/README.md @@ -7,22 +7,20 @@ repo-only sync workflows. - `internal-gateway-idea`: owns substantive idea definition, critical challenge, and retained planning before execution. -- `internal-gateway-review`: owns generic defect-first review for non-code and +- `internal-gateway-review-generic`: owns generic defect-first review for non-code and mixed artifacts before fixes. -- `internal-review-code`: owns dedicated code-focused review for source, tests, +- `internal-gateway-review-code`: owns dedicated code-focused review for source, tests, scripts, build metadata, dependency metadata, and code diffs. - `internal-gateway-critical-master`: owns pressure testing. -- `internal-gateway-simple-task`: owns concrete execution and approved compact - retained-plan consumption. +- `internal-gateway-simple-task`: owns concrete execution and approved retained-plan consumption. ## Active Gateway Agents | Agent | Use when | | --- | --- | | `internal-gateway-idea` | A vague idea or unresolved goal needs definition and retained planning. | -| `internal-gateway-review` | A concrete non-code or mixed artifact needs defect-first review before fixes. | -| `internal-review-code` | A concrete code target needs dedicated code review before merge or follow-up action. | +| `internal-gateway-review-generic` | A concrete non-code or mixed artifact needs defect-first review before fixes. | +| `internal-gateway-review-code` | A concrete code target needs dedicated code review before merge or follow-up action. | | `internal-gateway-critical-master` | A proposal or plan needs pressure before action. | | `internal-gateway-simple-task` | A concrete low-to-medium-risk task can finish through one focused lane. | - -Approved `extended` retained plans route directly to `internal-gateway-execute-plans`. +| `internal-gateway-execute-plans` | An approved retained plan under tmp/superpowers/plans/ is ready for execution. | diff --git a/.github/agents/internal-gateway-critical-master.agent.md b/.github/agents/internal-gateway-critical-master.agent.md index 28e5d1d2..f25fae00 100644 --- a/.github/agents/internal-gateway-critical-master.agent.md +++ b/.github/agents/internal-gateway-critical-master.agent.md @@ -11,3 +11,5 @@ agents: [] ## Core Skill - `internal-gateway-critical-master` +- Keep the full critical record internal and emit only the localized compact + card defined by the skill's output contract. diff --git a/.github/agents/internal-gateway-execute-plans.agent.md b/.github/agents/internal-gateway-execute-plans.agent.md new file mode 100644 index 00000000..3824ff1f --- /dev/null +++ b/.github/agents/internal-gateway-execute-plans.agent.md @@ -0,0 +1,13 @@ +--- +name: internal-gateway-execute-plans +description: "Use this agent when an approved retained plan under tmp/superpowers/plans/ is ready for execution or resume." +tools: ["read", "edit", "search", "execute"] +agents: [] +handoffs: [] +--- + +# Internal Gateway Execute Plans + +## Core Skill + +- `internal-gateway-execute-plans` diff --git a/.github/agents/internal-gateway-idea.agent.md b/.github/agents/internal-gateway-idea.agent.md index 8c4cc341..fa389f8c 100644 --- a/.github/agents/internal-gateway-idea.agent.md +++ b/.github/agents/internal-gateway-idea.agent.md @@ -2,20 +2,19 @@ name: internal-gateway-idea description: "Use this agent when a repository-owned request starts with a vague idea, unclear goal, unresolved option set, or needs substantive definition, convergence, critical challenge, and retained planning before execution." tools: ["read", "edit", "search", "execute", "web"] -disable-model-invocation: true agents: [] handoffs: - - label: "Next step: Execute compact plan" + - label: "Next step: Execute retained plan" agent: "internal-gateway-execute-plans" - prompt: "Execute only the approved compact retained plan left by the idea definition above. Verify the plan declares `Recommended consumer: internal-gateway-execute-plans`." + prompt: "Execute only the approved retained plan under tmp/superpowers/plans/ left by the idea definition above. Verify the plan path is exact and the user has approved execution." send: false - label: "Next step: Pressure-test decision" agent: "internal-gateway-critical-master" prompt: "Pressure-test the reasoning, assumptions, or failure modes behind the retained planning decision." send: false - label: "Next step: Review target" - agent: "internal-gateway-review" - prompt: "The work above became review-oriented rather than planning-oriented. Review the concrete artifact and stop before fixes." + agent: "internal-gateway-review-generic" + prompt: "The work above became review-oriented rather than planning-oriented. Use internal-gateway-review-generic to review the concrete artifact and stop before fixes." send: false --- diff --git a/.github/agents/internal-review-code.agent.md b/.github/agents/internal-gateway-review-code.agent.md similarity index 93% rename from .github/agents/internal-review-code.agent.md rename to .github/agents/internal-gateway-review-code.agent.md index 61925eff..b33b3053 100644 --- a/.github/agents/internal-review-code.agent.md +++ b/.github/agents/internal-gateway-review-code.agent.md @@ -1,5 +1,5 @@ --- -name: internal-review-code +name: internal-gateway-review-code description: "Senior repository code reviewer for source code, tests, scripts, build files, dependency files, and code-focused diffs before merge or follow-up action." tools: ["read", "search", "execute"] disable-model-invocation: true @@ -52,7 +52,7 @@ Evaluate every change across these five dimensions: - Resolve the concrete code target first: diff, pull request, changed file list, source file, test file, script, build file, dependency file, or generated-code boundary. - Read the spec, task description, or stated intent before judging implementation details when that evidence exists. - Review tests before implementation when tests are present because they reveal intended behavior and coverage gaps. -- Keep the review code-focused. Prefer `internal-gateway-review` when the primary target is an AI resource, workflow, policy, plan, documentation package, or mixed non-code artifact. +- Keep the review code-focused. Prefer `internal-gateway-review-generic` when the primary target is an AI resource, workflow, policy, plan, documentation package, or mixed non-code artifact. - Do not edit files, apply fixes, author plans, or route to peer agents. The user decides what to do after reading the report. - Every Critical, Important, and Suggestion finding must reference a concrete file path and line when line evidence is available. - If evidence is incomplete, mark the item as uncertain and recommend investigation instead of guessing. @@ -125,5 +125,5 @@ Categorize every finding: ## Composition - **Invoke directly when:** the user asks for a review of a specific code change, source file, test file, script, build file, dependency file, generated-code boundary, or pull request. -- **Prefer `internal-gateway-review` when:** the target is non-code, an AI resource, workflow, policy, plan, documentation package, or a mixed artifact where code is not the primary surface. +- **Prefer `internal-gateway-review-generic` when:** the target is non-code, an AI resource, workflow, policy, plan, documentation package, or a mixed artifact where code is not the primary surface. - **Do not invoke from another persona.** If deeper security, testing, or architecture ownership would change the decision, surface that as a recommendation in your report instead of delegating. diff --git a/.github/agents/internal-gateway-review.agent.md b/.github/agents/internal-gateway-review-generic.agent.md similarity index 94% rename from .github/agents/internal-gateway-review.agent.md rename to .github/agents/internal-gateway-review-generic.agent.md index f7c06f3f..825f44e1 100644 --- a/.github/agents/internal-gateway-review.agent.md +++ b/.github/agents/internal-gateway-review-generic.agent.md @@ -1,5 +1,5 @@ --- -name: internal-gateway-review +name: internal-gateway-review-generic description: "Use this agent when repository-owned work needs a defect-first review of a concrete non-code or mixed artifact, workflow, AI resource, policy, plan, bundle, or review package before acceptance or follow-up action." tools: ["read", "search", "execute"] disable-model-invocation: true @@ -49,7 +49,7 @@ Evaluate the target through the dimensions that apply to its surface: - **Plans and review packages:** retained plans, specs, audit packages, issue analysis, and decision-support reports. - **Mixed artifacts:** any target where code is secondary evidence inside a broader repository-owned artifact. -Prefer `internal-review-code` when the target is purely code: source, tests, scripts, build files, dependency files, generated-code boundaries, or a code-focused diff. +Prefer `internal-gateway-review-code` when the target is purely code: source, tests, scripts, build files, dependency files, generated-code boundaries, or a code-focused diff. ## Critical Counter-Analysis @@ -113,7 +113,7 @@ Use this report shape: - Use this agent when the user asks for review, audit, critique, merge-readiness assessment, prompt or agent review, workflow review, policy review, plan review, or artifact risk assessment. - Use this agent when the review target is not purely code or when the surface is mixed. -- Prefer `internal-review-code` when the requested review is specifically for source code, tests, scripts, build files, dependency files, or a code-focused diff. +- Prefer `internal-gateway-review-code` when the requested review is specifically for source code, tests, scripts, build files, dependency files, or a code-focused diff. - Do not use this agent when the user has already approved implementation, remediation, or execution. - Do not use this agent when there is no concrete review target; ask for the artifact, diff, file, PR, or package to review. - Do not delegate to peer agents or hand off to fix lanes. Name likely follow-up owners only as report context when that helps the user choose a next step. diff --git a/.github/agents/internal-gateway-simple-task.agent.md b/.github/agents/internal-gateway-simple-task.agent.md index 44db7a10..27ca1cfb 100644 --- a/.github/agents/internal-gateway-simple-task.agent.md +++ b/.github/agents/internal-gateway-simple-task.agent.md @@ -2,7 +2,6 @@ name: internal-gateway-simple-task description: "Use this agent when a concrete low-to-medium-risk repository-owned task can be answered, edited, diagnosed, or validated quickly in one bounded run, and should stop with reason when complexity or cost breaks that boundary." tools: ["read", "edit", "search", "execute", "web"] -disable-model-invocation: true agents: [] --- diff --git a/.github/agents/local-sync-external-resources.agent.md b/.github/agents/local-sync-external-resources.agent.md index b0a5efd6..71d73226 100644 --- a/.github/agents/local-sync-external-resources.agent.md +++ b/.github/agents/local-sync-external-resources.agent.md @@ -1,6 +1,6 @@ --- name: local-sync-external-resources -description: Use this agent when applying, auditing, or planning declared external resource refreshes through the staged sync CLI. +description: Use this agent when preparing, applying, auditing, or planning declared external resource refreshes through the staged sync CLI. tools: ["read", "edit", "search", "execute", "web"] disable-model-invocation: true agents: [] @@ -10,36 +10,50 @@ agents: [] ## Role -You are the manifest-driven sync wrapper for this repository's declared -external resource refreshes. Use this agent for audit, plan, and apply -operations through the single canonical CLI. +You are the decision and safety layer for this repository's declared external +resource refreshes. Use this agent to resolve operator intent, select the sync +mode, enforce authorization boundaries, and supervise the result. + +The paired core skill owns the reusable CLI procedure, mode mechanics, +validation sequence, and output schema. Treat it as the canonical operational +contract instead of reproducing its procedure here. ## Core Skill - `local-agent-sync-external-resources` -## Boundary - -- `sync` means `apply` by default unless the user explicitly asks for `audit` or `plan`. -- Refuse `apply` when a managed target has uncommitted changes unless the user supplies `--allow-dirty`. -- Stage all fetched and transformed resources outside the repository. -- Do not modify repository targets until the complete candidate tree, normalizations, overrides, and generated patch pass validation. -- Do not refresh or modify any imported skill while implementing sync tooling. - -## Safety - -- Workspace must live under `tmp/sync-externals-skills/`. -- Dirty managed targets block `apply` unless `--allow-dirty` is supplied. -- Override replay is atomic: if any override fails, no candidate changes reach the repository. - -## Completion Output - -In `Outcome`, include: - -- `Mode`: `apply`, `audit`, or `plan`. -- `Workspace`: external staging path or `n/a`. -- `Managed assets`: count from the manifest. -- `Changed paths`: list of repository-relative paths that changed. -- `Override results`: status of each replayed override. -- `Validation`: commands run and remaining gaps. -- `Blockers`: unresolved issues that prevent or narrow `apply`. +## Decision Contract + +- Load and follow the core skill for every in-scope operation. +- Preserve the requested mode. Bare `sync` means `apply`; never promote + `audit` or `plan` into a mutating or networked mode. +- Run networked `prepare` only when the user explicitly requests source + preparation or otherwise authorizes that network step. +- If an offline mode lacks prepared source metadata, report the blocker and the + required `prepare` action. Do not fetch automatically. +- Treat `--allow-dirty` as explicit risk acceptance. Never infer it from a + general request to continue. +- Use the core skill's single public CLI and declared manifest. Do not + reconstruct sync logic with ad hoc Git, file-copy, or package-manager + commands. + +## Boundary And Stop Conditions + +- Stay in this lane only for declared external-resource `prepare`, `audit`, + `plan`, and `apply` work. +- Stop when the request would change manifest scope, source pins, override + policy, or imported content outside the staged sync contract. +- Stop before `apply` when managed targets are dirty, candidate validation + fails, an override cannot replay, or the external workspace requirement is + not met. +- Keep live benchmarking separate and require its dedicated authorization. +- When changing the sync tooling itself, do not refresh or modify imported + resources in the same task. + +## Outcome + +Report the core skill's complete result without optimistic compression: +selected mode, workspace, source root when used, managed asset count, changed +paths, per-source metrics, override results, validation evidence, and blockers. +State why the selected mode was authorized and name any exact user action +required before a blocked network or mutation step. diff --git a/.github/agents/local-sync-global-copilot-configs-into-repo.agent.md b/.github/agents/local-sync-global-copilot-configs-into-repo.agent.md deleted file mode 100644 index 8f6cfb02..00000000 --- a/.github/agents/local-sync-global-copilot-configs-into-repo.agent.md +++ /dev/null @@ -1,55 +0,0 @@ ---- -name: local-sync-global-copilot-configs-into-repo -description: Use this agent when planning, auditing, or applying consumer-repository alignment to this repository's managed GitHub Copilot baseline and explicitly shared hygiene files while preserving target `local-*` assets. -tools: ["read", "edit", "search", "execute", "web"] -disable-model-invocation: true -agents: [] ---- - -# Local Sync Global Copilot Configs Into Repo - -## Role - -You are the cross-repository baseline propagation owner for this repository's managed GitHub Copilot assets. - -Use this agent for route selection, mode selection, approval posture, and boundary decisions. The paired core skill owns the reusable analyze, plan, apply, plan-file, automation, mirrored-scope, and reporting procedure. - -## Core Skill - -- `local-agent-sync-global-copilot-configs-into-repo` - -## Routing Rules - -- Use this agent for consumer-repository baseline propagation, drift assessment, `plan`, `audit`, and explicit `apply` runs. -- Use this agent when the target repository must inherit the root `AGENTS.md` policy model, review-only Copilot configuration, `.github/INVENTORY.md`, repository-root `LESSONS_LEARNED.md`, and explicitly shared hygiene files. -- Select `apply` only on explicit request, after the current evidence shows a conflict-safe plan and no unmanaged target-local cleanup is being implied. -- Do not use this agent for source-side catalog governance, external-resource refreshes, or managed-scope redesign in this repository; recommend `local-sync-external-resources` or `internal-gateway-idea` as appropriate. -- Do not use this agent for one-resource agent or skill authoring; recommend `internal-agent-creator` or `internal-skill-creator` as appropriate. -- When current platform behavior decides sync policy, validate it through `internal-copilot-docs-research` before changing the contract. - -## Boundary Definition - -- Stay in this lane while the task is consumer-repository baseline propagation, drift assessment, or sync `plan`/`audit`/`apply` work. -- Preserve target `local-*` assets unless the user explicitly approves target-local cleanup. -- Mirror only the managed cross-repository baseline declared by the core skill; do not expand source-managed scope from this agent. -- If the request is really source-side catalog governance, source-side redesign, or a local edit outside the sync lane, explain the mismatch and recommend the better owner visibly. -- Do not route, dispatch, or delegate from this lane. - -## Core Rules - -- Treat this repository as the source of truth for the managed sync baseline. -- Keep root guidance layered: `AGENTS.md` is the agent policy entrypoint, `.github/copilot-instructions.md` is review-only for GitHub.com Copilot code review, and `.github/INVENTORY.md` is the live catalog. -- Keep target assumptions narrow and let the core skill own mirrored categories, exclusions, automation entrypoints, plan-file lifecycle, and validation sequence. -- When repository-root `LESSONS_LEARNED.md` is in scope, preserve or migrate target-authored lesson rows through the core skill workflow. -- When the source baseline includes approved imported-asset override registries or replay patches, mirror them as source-managed governance assets rather than creating target-local hidden forks. -- Require explicit approval before deleting or rewriting target-owned content outside the managed baseline. - -## Output Expectations - -- Target repository, selected mode, and why that mode is valid. -- Source baseline and target evidence used for the decision. -- Root-guidance alignment strategy and `LESSONS_LEARNED.md` sync status. -- Preserved `local-*` assets and any approved target-only cleanup. -- Boundary or approval decisions that affected the selected mode. -- Validation results, remaining blockers, and explicit validation gaps. -- Used agents, instructions, skills, and other resources when a narrower completion-report contract requires that detail. diff --git a/.github/agents/local-sync-install-ai-resources.agent.md b/.github/agents/local-sync-install-ai-resources.agent.md index e9501289..a536baa3 100644 --- a/.github/agents/local-sync-install-ai-resources.agent.md +++ b/.github/agents/local-sync-install-ai-resources.agent.md @@ -14,4 +14,4 @@ agents: [] When the user names `agents.md`, `agents.md` means `sync --targets agents.md`. This updates `~/.agents/AGENTS.md` from root `AGENTS.md` without the -`` block. +repository-local `AGENTS.local.md` file. diff --git a/.github/agents/local-sync-repos.agent.md b/.github/agents/local-sync-repos.agent.md new file mode 100644 index 00000000..6245c59c --- /dev/null +++ b/.github/agents/local-sync-repos.agent.md @@ -0,0 +1,49 @@ +--- +name: local-sync-repos +description: Use this agent when planning, auditing, or applying consumer-repository alignment to this repository's managed instruction, root-policy, and shared-hygiene baseline while preserving target local-* assets. +tools: ["read", "edit", "search", "execute"] +disable-model-invocation: true +agents: [] +--- + +# Local Sync Repos + +## Role + +You are the cross-repository baseline propagation owner for this repository's managed instruction, root-policy, and shared-hygiene files. + +Use this agent for route selection, mode selection, approval posture, and boundary decisions. The paired core skill owns the reusable plan, apply, reporting, and automation procedure. + +## Core Skill + +- `local-sync-repos` + +## Routing Rules + +- Use this agent for consumer-repository baseline propagation, drift assessment, `plan`, and explicit `apply` runs. +- Select `apply` only on explicit request, after the current evidence shows a conflict-safe plan and no unmanaged target-local cleanup is being implied. +- Do not use this agent for source-side catalog governance, external-resource refreshes, home AI sync, or managed-scope redesign; recommend `local-sync-external-resources`, `local-agent-sync-install-ai-resources`, or `internal-gateway-idea` as appropriate. + +## Boundary Definition + +- Stay in this lane while the task is consumer-repository baseline propagation, drift assessment, or sync `plan`/`apply` work. +- Manage only the eight approved target path categories: `AGENTS.md`, `.python-version`, `.pre-commit-config.yaml`, `.editorconfig`, `.github/copilot-instructions.md`, `.github/workflows/_pre-commit.yml`, `.github/instructions/**`, and `AGENTS.local.md`. +- Do not synchronize agents, skills, prompts, inventory, documentation, lesson ledgers, VS Code settings, or unrelated workflows. +- Preserve target `local-*` instruction files unless the user explicitly approves target-local cleanup. +- If the request is really source-side catalog governance, source-side redesign, or a local edit outside the sync lane, explain the mismatch and recommend the better owner visibly. +- Do not route, dispatch, or delegate from this lane. + +## Stop Conditions + +- Source managed files are missing. +- A dirty target overlaps a planned managed mutation. +- A saved plan fingerprint is missing or stale. +- The user has not explicitly approved `apply`. + +## Output Expectations + +- Target repository, selected mode, and why that mode is valid. +- Source baseline and target evidence used for the decision. +- Preserved `local-*` assets and any approved target-only cleanup. +- Boundary or approval decisions that affected the selected mode. +- Validation results, remaining blockers, and explicit validation gaps. diff --git a/.github/dependabot.yml b/.github/dependabot.yml index bc1226b7..7b23352b 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -42,24 +42,24 @@ updates: update-types: - "minor" - "patch" - - package-ecosystem: "pre-commit" - directory: "/" - schedule: - interval: "weekly" - day: "monday" - time: "09:00" - timezone: "Europe/Rome" - open-pull-requests-limit: 5 - labels: - - "dependencies" - - "pre-commit" - commit-message: - prefix: "deps" - include: "scope" - groups: - pre-commit-minor-patch: - patterns: - - "*" - update-types: - - "minor" - - "patch" + # - package-ecosystem: "pre-commit" + # directory: "/" + # schedule: + # interval: "weekly" + # day: "monday" + # time: "09:00" + # timezone: "Europe/Rome" + # open-pull-requests-limit: 5 + # labels: + # - "dependencies" + # - "pre-commit" + # commit-message: + # prefix: "deps" + # include: "scope" + # groups: + # pre-commit-minor-patch: + # patterns: + # - "*" + # update-types: + # - "minor" + # - "patch" diff --git a/.github/prompts/internal-mega-review.prompt.md b/.github/prompts/internal-mega-review.prompt.md index aba427ed..280e60d0 100644 --- a/.github/prompts/internal-mega-review.prompt.md +++ b/.github/prompts/internal-mega-review.prompt.md @@ -1,6 +1,6 @@ --- name: "internal-mega-review" -agent: "internal-gateway-review" +agent: "internal-gateway-review-generic" description: "Run a complete advisor-only mega review for one or more repositories and write split English analysis under each repo tmp/" argument-hint: "Repository paths or names; optional focus and constraints; retained output is English" --- @@ -26,7 +26,7 @@ Use these standards-repository sources first for governance, evidence, and split - [AGENTS.md](../../AGENTS.md) - [.github/copilot-instructions.md](../copilot-instructions.md) - [.github/INVENTORY.md](../INVENTORY.md) -- [.github/agents/internal-gateway-review.agent.md](../agents/internal-gateway-review.agent.md) +- [.github/agents/internal-gateway-review-generic.agent.md](../agents/internal-gateway-review-generic.agent.md) - [.github/agents/internal-gateway-critical-master.agent.md](../agents/internal-gateway-critical-master.agent.md) - [.github/skills/internal-gateway-writing-plans/SKILL.md](../skills/internal-gateway-writing-plans/SKILL.md) diff --git a/.github/prompts/internal-review-ai-resources.prompt.md b/.github/prompts/internal-review-ai-resources.prompt.md index 28fbeb82..dbaed877 100644 --- a/.github/prompts/internal-review-ai-resources.prompt.md +++ b/.github/prompts/internal-review-ai-resources.prompt.md @@ -1,6 +1,6 @@ --- name: "internal-review-ai-resources" -agent: "internal-gateway-review" +agent: "internal-gateway-review-generic" description: "Review repository-owned AI resources, referenced assets, and flow behavior across AGENTS.md and .github" argument-hint: "Target one file, one or more folders, the full AI catalog, or an existing retained report package" --- @@ -34,7 +34,7 @@ Use these repository sources first: - [.github/copilot-instructions.md](../copilot-instructions.md) - [.github/INVENTORY.md](../INVENTORY.md) - [INTERNAL_CONTRACT.md](../../INTERNAL_CONTRACT.md) -- [.github/agents/internal-gateway-review.agent.md](../agents/internal-gateway-review.agent.md) +- [.github/agents/internal-gateway-review-generic.agent.md](../agents/internal-gateway-review-generic.agent.md) - [.github/skills/internal-review-ai-resources/SKILL.md](../skills/internal-review-ai-resources/SKILL.md) Then use `internal-review-ai-resources` as the reusable qualitative owner for diff --git a/.github/prompts/internal-sync-plan.prompt.md b/.github/prompts/internal-sync-plan.prompt.md deleted file mode 100644 index 6de367fc..00000000 --- a/.github/prompts/internal-sync-plan.prompt.md +++ /dev/null @@ -1,48 +0,0 @@ ---- -name: "internal-sync-plan" -agent: "local-sync-global-copilot-configs-into-repo" -description: "Plan a source-authoritative consumer-repository sync without applying it." -argument-hint: "Source repo or branch, target repo, requested sync change, and target-local exceptions to preserve" ---- - - - -Source repository or branch: -${input:source:Describe the source standards repo or branch under consideration.} - -Target repository or consumer scope: -${input:target:Describe the consumer repo, branch, or sync target.} - -Requested sync or governance change: -${input:change:Describe the desired sync, refresh, retire, or drift-remediation action.} - -Target-local exceptions to preserve: -${input:local:List local assets, overrides, or no-touch areas if known.} - -Use these sources first: - -- [AGENTS.md](../../AGENTS.md) -- [.github/copilot-instructions.md](../copilot-instructions.md) -- [.github/INVENTORY.md](../INVENTORY.md) -- [.github/agents/local-sync-global-copilot-configs-into-repo.agent.md](../agents/local-sync-global-copilot-configs-into-repo.agent.md) -- [.github/skills/local-agent-sync-global-copilot-configs-into-repo/SKILL.md](../skills/local-agent-sync-global-copilot-configs-into-repo/SKILL.md) -- [VERSION](../../VERSION) - -Planning contract: - -- This prompt is for source-authoritative consumer-repository sync planning, not - for source-side catalog governance apply work. -- If the real task is source-side `.github/` catalog governance, route it to - `local-sync-external-resources` instead of forcing this plan shape. -- Keep target `local-*` assets and any consumer-local override layer visible in - the plan. -- Prefer a conflict-safe plan with explicit validation over implied apply - approval. - -Produce a sync brief with: - -1. selected mode and why it fits -2. source-managed scope versus target-local assets -3. planned create, update, preserve, and delete actions -4. validation path before apply -5. blockers, rollout risks, and no-apply conditions diff --git a/.github/repo-profiles.yml b/.github/repo-profiles.yml index 5ef230c1..cd2b3b8e 100644 --- a/.github/repo-profiles.yml +++ b/.github/repo-profiles.yml @@ -61,10 +61,10 @@ profiles: recommended_skills: - skills/internal-markdown/SKILL.md - skills/internal-yaml/SKILL.md - - skills/openai-docx/SKILL.md - - skills/openai-pdf/SKILL.md - - skills/openai-slides/SKILL.md - - skills/openai-spreadsheet/SKILL.md + - skills/anthropic-docx/SKILL.md + - skills/anthropic-pdf/SKILL.md + - skills/anthropic-pptx/SKILL.md + - skills/anthropic-xlsx/SKILL.md infrastructure-heavy: description: Infrastructure repositories with Terraform as primary language. @@ -80,7 +80,7 @@ profiles: - skills/internal-terraform/SKILL.md - skills/internal-yaml/SKILL.md - skills/internal-markdown/SKILL.md - - skills/internal-azure-strategic/SKILL.md + - skills/internal-azure/SKILL.md - skills/internal-azure-organization-structure/SKILL.md - skills/internal-azure-governance/SKILL.md - skills/internal-azure-operations/SKILL.md diff --git a/.github/scripts/lib/internal_skills.py b/.github/scripts/lib/internal_skills.py index 4b002a02..a54e6a63 100644 --- a/.github/scripts/lib/internal_skills.py +++ b/.github/scripts/lib/internal_skills.py @@ -23,6 +23,7 @@ INLINE_TEMPLATE_THRESHOLD = 4 TRIGGER_FIRST_PREFIXES = ( "Use when", + "Use only when", "Use first when", "Use this", "Use before", @@ -240,14 +241,17 @@ def validate_openai_yaml(skill_dir: Path, skill_name: str) -> list[Finding]: suggestion="Add a deterministic default prompt that shows how to invoke the skill.", ) ) - elif f"${skill_name}" not in default_prompt: + elif f"${skill_name}" not in default_prompt and f"/{skill_name}" not in default_prompt: findings.append( Finding( severity="non-blocking", code="default-prompt-skill-mention", path=openai_yaml.as_posix(), message="interface.default_prompt does not mention the skill identifier explicitly.", - suggestion=f"Mention ${skill_name} in the default prompt for consistent invocation hints.", + suggestion=( + f"Mention ${skill_name} or /{skill_name} in the default prompt " + "for consistent invocation hints." + ), ) ) diff --git a/.github/scripts/lib/inventory.py b/.github/scripts/lib/inventory.py index 0fef6980..21d1877c 100644 --- a/.github/scripts/lib/inventory.py +++ b/.github/scripts/lib/inventory.py @@ -18,11 +18,11 @@ ".github/scripts/*.sh", ".github/scripts/lib/*.py", ) -OFFICE_SUPPORT_ONLY_SKILLS = ( - ".github/skills/openai-docx/SKILL.md", - ".github/skills/openai-pdf/SKILL.md", - ".github/skills/openai-slides/SKILL.md", - ".github/skills/openai-spreadsheet/SKILL.md", +DOCUMENT_SUPPORT_ONLY_SKILLS = ( + ".github/skills/anthropic-docx/SKILL.md", + ".github/skills/anthropic-pdf/SKILL.md", + ".github/skills/anthropic-pptx/SKILL.md", + ".github/skills/anthropic-xlsx/SKILL.md", ) IGNORED_SCRIPT_BASENAMES = {"__init__.py"} @@ -81,13 +81,13 @@ def render_inventory_markdown(sections: dict[str, list[str]]) -> str: if entries: lines.extend(f"- `{entry}`" for entry in entries) if section == "Skills": - office_entries = [entry for entry in entries if entry in OFFICE_SUPPORT_ONLY_SKILLS] - if office_entries: + doc_entries = [entry for entry in entries if entry in DOCUMENT_SUPPORT_ONLY_SKILLS] + if doc_entries: lines.append("") - lines.append("### Support-only imported office skills") + lines.append("### Support-only imported document skills") lines.append("") lines.append( - "These imported `openai-*` office skills remain support-only depth for repositories that explicitly need document workflows." + "These vendor-prefixed imported document skills remain support-only depth for repositories that explicitly need document workflows." ) else: lines.append(EMPTY_MESSAGES[section]) diff --git a/.github/scripts/lib/syncing.py b/.github/scripts/lib/syncing.py deleted file mode 100644 index e8e72c9c..00000000 --- a/.github/scripts/lib/syncing.py +++ /dev/null @@ -1,1046 +0,0 @@ -from __future__ import annotations - -import hashlib -import json -import re -from dataclasses import dataclass -from datetime import datetime, timezone -from pathlib import Path -from shutil import copy2 - -from .fingerprinting import HASH_ALGO, NORMALIZATION_VERSION, build_fingerprint -from .inventory import render_inventory_markdown, sections_from_catalog_paths -from .jsonc import JsoncParseError, apply_managed_vscode_copilot_settings -from .shared import ( - ARCHITECTURE_PATH, - CONSUMER_LOCAL_KNOWLEDGE_TEMPLATES, - DOCS_README_PATH, - INVENTORY_PATH, - LEGACY_ARCHITECTURE_PATH, - LEGACY_LOCAL_ARCHITECTURE_PATH, - LEGACY_LOCAL_REPOSITORY_CONTEXT_PATH, - LEGACY_REPOSITORY_CONTEXT_PATH, - LEGACY_RUNTIME_FIT_PATH, - LESSONS_PATH, - MANAGED_ROOT_FILES, - MANAGED_WORKFLOW_FILES, - REPOSITORY_CONTEXT_PATH, - RETIRED_RUNTIME_OPERATING_MODEL_PATH, - STRUCTURE_PATH, - TECH_PATH, - VSCODE_COPILOT_SETTINGS, - VSCODE_SETTINGS_PATH, - SyncOperation, - SyncPlan, - action_sort_key, - all_files_under, - git_dirty_paths, - git_revision, - is_consumer_sync_excluded_path, - is_git_dirty, - is_ignored_sync_path, - is_local_asset, - read_text, - sha256_file, - write_text, -) - -MANAGED_SKILL_DIR = ".github/skills" -SYNC_PLAN_PATH = "tmp/copilot-sync.plan.md" -SYNC_MANIFEST_PATH = ".github/copilot-sync.manifest.json" -VERSION_PATH = "VERSION" -LEGACY_SYNC_ARTIFACT_PATHS = ( - "tmp/internal-sync-copilot-configs.plan.md", - ".github/internal-sync-copilot-configs.manifest.json", -) -TARGET_GITIGNORE_PATH = ".gitignore" -TARGET_SUPERPOWERS_IGNORE_ENTRY = "/tmp/superpowers/" -DEFAULT_NO_PENDING_LESSONS_MARKER = "No pending lessons currently require codification." -NO_PENDING_LESSONS_MARKERS = { - "No pending lessons currently.", - DEFAULT_NO_PENDING_LESSONS_MARKER, -} - -ARCHITECTURE_LEGACY_PATHS = ( - LEGACY_LOCAL_ARCHITECTURE_PATH, - LEGACY_ARCHITECTURE_PATH, -) -REPOSITORY_CONTEXT_LEGACY_PATHS = ( - LEGACY_LOCAL_REPOSITORY_CONTEXT_PATH, - LEGACY_REPOSITORY_CONTEXT_PATH, -) - - -@dataclass(frozen=True) -class PendingLessonsTable: - column_count: int - data_start: int - data_end: int - section_end: int - - -@dataclass(frozen=True) -class ConsumerLocalKnowledgeSpec: - path: str - preserve_reason: str - create_reason: str - legacy_paths: tuple[str, ...] = () - - -CONSUMER_LOCAL_KNOWLEDGE_SPECS = ( - ConsumerLocalKnowledgeSpec( - path=DOCS_README_PATH, - preserve_reason="Preserved consumer-local docs guide after scaffold materialization.", - create_reason="Consumer-local docs guide missing; create scaffold from the source template.", - ), - ConsumerLocalKnowledgeSpec( - path=ARCHITECTURE_PATH, - preserve_reason="Preserved consumer-local architecture contract after scaffold materialization.", - create_reason="Consumer-local architecture contract missing; create scaffold from the source template.", - legacy_paths=ARCHITECTURE_LEGACY_PATHS, - ), - ConsumerLocalKnowledgeSpec( - path=REPOSITORY_CONTEXT_PATH, - preserve_reason="Preserved consumer-local repository context after scaffold materialization.", - create_reason="Consumer-local repository context missing; create scaffold from the source template.", - legacy_paths=REPOSITORY_CONTEXT_LEGACY_PATHS, - ), - ConsumerLocalKnowledgeSpec( - path=TECH_PATH, - preserve_reason="Preserved consumer-local technology document after scaffold materialization.", - create_reason="Consumer-local technology document missing; create scaffold from the source template.", - ), - ConsumerLocalKnowledgeSpec( - path=STRUCTURE_PATH, - preserve_reason="Preserved consumer-local structure document after scaffold materialization.", - create_reason="Consumer-local structure document missing; create scaffold from the source template.", - ), -) - - -def append_vscode_settings_operation( - target_root: Path, - operations: list[SyncOperation], -) -> None: - target_path = target_root / VSCODE_SETTINGS_PATH - if not target_path.exists(): - desired_content = render_minimal_vscode_settings_jsonc() - operations.append( - SyncOperation( - action="ensure", - path=VSCODE_SETTINGS_PATH, - reason=( - "Target VS Code settings file is missing; create it with managed Copilot-only settings." - ), - source_hash=sha256_text(desired_content), - target_hash=None, - ) - ) - return - - existing = read_text(target_path) - try: - updated = apply_managed_vscode_copilot_settings(existing) - except JsoncParseError as error: - operations.append( - SyncOperation( - action="manual", - path=VSCODE_SETTINGS_PATH, - reason=f"VS Code settings require manual reconciliation: {error}", - source_hash=None, - target_hash=sha256_file(target_path), - ) - ) - return - - if updated == existing: - operations.append( - SyncOperation( - action="unchanged", - path=VSCODE_SETTINGS_PATH, - reason="Target VS Code Copilot settings already match the managed field-level contract.", - source_hash=sha256_text(updated), - target_hash=sha256_file(target_path), - ) - ) - return - - operations.append( - SyncOperation( - action="ensure", - path=VSCODE_SETTINGS_PATH, - reason="Target VS Code settings must be merged to enforce managed Copilot-only values.", - source_hash=sha256_text(updated), - target_hash=sha256_file(target_path), - ) - ) - - -def render_minimal_vscode_settings_jsonc() -> str: - return ( - "{\n" - ' "github.copilot.chat.codeGeneration.useInstructionFiles": false,\n' - ' "chat.instructionsFilesLocations": {\n' - ' ".github/instructions": false\n' - " }\n" - "}\n" - ) - - -def build_sync_plan(source_root: Path, target_root: Path) -> SyncPlan: - source_root = source_root.resolve() - target_root = target_root.resolve() - source_version = read_source_version(source_root) - target_manifest_source_version = read_target_manifest_source_version(target_root) - - source_files = discover_source_sync_files(source_root) - target_files = discover_target_managed_files(target_root) - target_excluded_files = discover_target_excluded_sync_files(target_root) - operations: list[SyncOperation] = [] - local_assets: list[str] = [] - generated_lessons: str | None = None - - append_consumer_local_knowledge_operations( - source_root=source_root, - target_root=target_root, - operations=operations, - local_assets=local_assets, - ) - append_vscode_settings_operation(target_root=target_root, operations=operations) - - for relative_path in sorted(source_files): - source_path = source_root / relative_path - target_path = target_root / relative_path - if relative_path == LESSONS_PATH: - generated_lessons = render_synced_lessons( - read_text(source_path), - read_text(target_path) if target_path.exists() else None, - ) - desired_hash = sha256_text(generated_lessons) - if not target_path.exists(): - operations.append( - SyncOperation( - action="create", - path=relative_path, - reason="Target learning ledger missing; create it from the source structure.", - source_hash=desired_hash, - target_hash=None, - ) - ) - continue - - target_hash = sha256_file(target_path) - action = "unchanged" if target_hash == desired_hash else "update" - reason = ( - "Target learning ledger already matches the source structure and preserved lessons." - if action == "unchanged" - else "Target learning ledger must align with the source structure while preserving target-authored lessons." - ) - operations.append( - SyncOperation( - action=action, - path=relative_path, - reason=reason, - source_hash=desired_hash, - target_hash=target_hash, - ) - ) - continue - - source_hash = sha256_file(source_path) - if not target_path.exists(): - operations.append( - SyncOperation( - action="create", - path=relative_path, - reason="Source-managed file missing from target.", - source_hash=source_hash, - target_hash=None, - ) - ) - continue - - target_hash = sha256_file(target_path) - action = "unchanged" if source_hash == target_hash else "update" - reason = ( - "Already aligned with source." - if action == "unchanged" - else "Target file differs from source." - ) - operations.append( - SyncOperation( - action=action, - path=relative_path, - reason=reason, - source_hash=source_hash, - target_hash=target_hash, - ) - ) - - for relative_path in sorted(target_files - source_files - {INVENTORY_PATH}): - if is_local_asset(relative_path): - local_assets.append(relative_path) - operations.append( - SyncOperation( - action="preserve", - path=relative_path, - reason="Preserved target-owned local extension.", - source_hash=None, - target_hash=sha256_file(target_root / relative_path), - ) - ) - continue - operations.append( - SyncOperation( - action="delete", - path=relative_path, - reason="Target-only non-local asset inside a source-managed category.", - source_hash=None, - target_hash=sha256_file(target_root / relative_path), - ) - ) - - for relative_path in sorted(target_excluded_files): - operations.append( - SyncOperation( - action="delete", - path=relative_path, - reason="Target internal-sync resource must be removed from consumer repositories.", - source_hash=None, - target_hash=sha256_file(target_root / relative_path), - ) - ) - - for relative_path in LEGACY_SYNC_ARTIFACT_PATHS: - legacy_path = target_root / relative_path - if not legacy_path.exists(): - continue - operations.append( - SyncOperation( - action="delete", - path=relative_path, - reason="Legacy internal-sync tracking artifact must be removed from consumer repositories.", - source_hash=None, - target_hash=sha256_file(legacy_path), - ) - ) - - retired_runtime_documents = { - LEGACY_RUNTIME_FIT_PATH: ( - "Legacy runtime-fit document is retired; runtime workflow guidance " - "now lives in root guidance and skills." - ), - RETIRED_RUNTIME_OPERATING_MODEL_PATH: ( - "Retired source-managed runtime operating model document; runtime " - "workflow guidance now lives in root guidance and skills." - ), - } - for relative_path, reason in retired_runtime_documents.items(): - retired_runtime_path = target_root / relative_path - if not retired_runtime_path.exists(): - continue - operations.append( - SyncOperation( - action="delete", - path=relative_path, - reason=reason, - source_hash=None, - target_hash=sha256_file(retired_runtime_path), - ) - ) - - future_inventory_paths = sorted( - catalog_path - for catalog_path in source_files - if catalog_path.startswith( - ( - ".github/agents/", - ".github/instructions/", - ".github/prompts/", - ".github/skills/", - ) - ) - ) - future_inventory_paths.extend( - catalog_path - for catalog_path in local_assets - if catalog_path.startswith( - ( - ".github/agents/", - ".github/instructions/", - ".github/prompts/", - ".github/skills/", - ) - ) - ) - generated_inventory = render_inventory_markdown( - sections_from_catalog_paths(future_inventory_paths) - ) - - inventory_path = target_root / INVENTORY_PATH - current_inventory = read_text(inventory_path) if inventory_path.exists() else None - inventory_action = ( - "unchanged" if current_inventory == generated_inventory else "rebuild" - ) - inventory_reason = ( - "Inventory already reflects target state." - if inventory_action == "unchanged" - else "Inventory must be rebuilt from target filesystem state." - ) - operations.append( - SyncOperation( - action=inventory_action, - path=INVENTORY_PATH, - reason=inventory_reason, - source_hash=None, - target_hash=( - sha256_file(inventory_path) if inventory_path.exists() else None - ), - ) - ) - - gitignore_path = target_root / TARGET_GITIGNORE_PATH - current_gitignore = read_text(gitignore_path) if gitignore_path.exists() else None - generated_gitignore = ensure_superpowers_gitignore_entry(current_gitignore) - gitignore_action = ( - "unchanged" if current_gitignore == generated_gitignore else "ensure" - ) - gitignore_reason = ( - "Target .gitignore already ignores tmp/superpowers." - if gitignore_action == "unchanged" - else "Target .gitignore must ignore tmp/superpowers for Superpowers-generated working artifacts." - ) - operations.append( - SyncOperation( - action=gitignore_action, - path=TARGET_GITIGNORE_PATH, - reason=gitignore_reason, - source_hash=None, - target_hash=( - sha256_file(gitignore_path) if gitignore_path.exists() else None - ), - ) - ) - - ordered_operations = tuple( - sorted( - operations, - key=lambda operation: (action_sort_key(operation.action), operation.path), - ) - ) - planned_paths = {operation.path for operation in ordered_operations} - dirty_paths = tuple( - path for path in git_dirty_paths(target_root) if path in planned_paths - ) - managed_mutation_paths = tuple( - sorted( - { - operation.path - for operation in ordered_operations - if operation.action - in { - "create", - "update", - "rename", - "ensure", - "rebuild", - "delete", - "manual", - } - } - ) - ) - dirty_managed_overlap = tuple( - path for path in dirty_paths if path in managed_mutation_paths - ) - return SyncPlan( - source_root=source_root, - target_root=target_root, - source_revision=git_revision(source_root), - source_version=source_version, - target_manifest_source_version=target_manifest_source_version, - target_dirty=is_git_dirty(target_root), - stacks=tuple(detect_target_stacks(target_root)), - operations=ordered_operations, - local_assets=tuple(sorted(local_assets)), - generated_inventory=generated_inventory, - generated_lessons=generated_lessons, - generated_gitignore=generated_gitignore, - dirty_paths=dirty_paths, - managed_mutation_paths=managed_mutation_paths, - dirty_managed_overlap=dirty_managed_overlap, - ) - - -def append_consumer_local_knowledge_operations( - source_root: Path, - target_root: Path, - operations: list[SyncOperation], - local_assets: list[str], -) -> None: - for spec in CONSUMER_LOCAL_KNOWLEDGE_SPECS: - _append_consumer_knowledge_operation( - source_root=source_root, - target_root=target_root, - operations=operations, - local_assets=local_assets, - spec=spec, - ) - - -def _append_consumer_knowledge_operation( - source_root: Path, - target_root: Path, - operations: list[SyncOperation], - local_assets: list[str], - spec: ConsumerLocalKnowledgeSpec, -) -> None: - template_path = source_root / CONSUMER_LOCAL_KNOWLEDGE_TEMPLATES[spec.path] - target_path = target_root / spec.path - source_hash = sha256_file(template_path) if template_path.exists() else None - legacy_paths = [ - legacy_path - for legacy_path in spec.legacy_paths - if (target_root / legacy_path).exists() - ] - - if target_path.exists() and legacy_paths: - local_assets.append(spec.path) - operations.append( - SyncOperation( - action="manual", - path=spec.path, - reason=( - f"Both canonical {spec.path} and legacy {', '.join(legacy_paths)} exist; reconcile manually before apply." - ), - source_hash=source_hash, - target_hash=sha256_file(target_path), - ) - ) - return - - if target_path.exists(): - local_assets.append(spec.path) - operations.append( - SyncOperation( - action="preserve", - path=spec.path, - reason=spec.preserve_reason, - source_hash=source_hash, - target_hash=sha256_file(target_path), - ) - ) - return - - if len(legacy_paths) > 1: - operations.append( - SyncOperation( - action="manual", - path=spec.path, - reason=( - f"Multiple legacy paths for {spec.path} exist ({', '.join(legacy_paths)}); reconcile manually before apply." - ), - source_hash=source_hash, - target_hash=sha256_file(target_root / legacy_paths[0]), - ) - ) - return - - if legacy_paths: - legacy_path = legacy_paths[0] - operations.append( - SyncOperation( - action="rename", - path=spec.path, - reason=f"Legacy {legacy_path} should move to consumer-local {spec.path}.", - source_hash=None, - target_hash=sha256_file(target_root / legacy_path), - ) - ) - return - - if template_path.exists(): - operations.append( - SyncOperation( - action="create", - path=spec.path, - reason=spec.create_reason, - source_hash=source_hash, - target_hash=None, - ) - ) - - -def render_synced_lessons(source_content: str, target_content: str | None) -> str: - source_lines = source_content.splitlines() - pending_table = find_pending_lessons_table(source_lines) - if pending_table is None: - return ensure_trailing_newline(source_content) - - target_rows = extract_pending_lessons_rows(target_content) - normalized_rows = [ - normalize_pending_lessons_row(row, pending_table.column_count) - for row in target_rows - ] - section_suffix = [ - line - for line in source_lines[pending_table.data_end : pending_table.section_end] - if line.strip() not in NO_PENDING_LESSONS_MARKERS - ] - pending_section_lines = [format_markdown_table_row(row) for row in normalized_rows] - if not normalized_rows: - marker = ( - find_no_pending_lessons_marker(target_content) - or find_no_pending_lessons_marker(source_content) - or DEFAULT_NO_PENDING_LESSONS_MARKER - ) - if not section_suffix or section_suffix[-1].strip(): - section_suffix.append("") - section_suffix.append(marker) - - merged_lines = ( - source_lines[: pending_table.data_start] - + pending_section_lines - + section_suffix - + source_lines[pending_table.section_end :] - ) - return ensure_trailing_newline("\n".join(merged_lines)) - - -def find_no_pending_lessons_marker(content: str | None) -> str | None: - if not content: - return None - for line in content.splitlines(): - stripped = line.strip() - if stripped in NO_PENDING_LESSONS_MARKERS: - return stripped - return None - - -def extract_pending_lessons_rows(content: str | None) -> list[list[str]]: - if not content: - return [] - lines = content.splitlines() - pending_table = find_pending_lessons_table(lines) - if pending_table is None: - return [] - - rows: list[list[str]] = [] - for line in lines[pending_table.data_start : pending_table.section_end]: - stripped = line.strip() - if not stripped: - continue - if not line.lstrip().startswith("|"): - break - cells = parse_markdown_table_row(line) - if any(cell for cell in cells): - rows.append(cells) - return rows - - -def find_pending_lessons_table(lines: list[str]) -> PendingLessonsTable | None: - section_start: int | None = None - for index, line in enumerate(lines): - if line.strip() == "## Pending Rules": - section_start = index + 1 - break - if section_start is None: - return None - - section_end = len(lines) - for index in range(section_start, len(lines)): - if lines[index].startswith("## "): - section_end = index - break - - header_index: int | None = None - for index in range(section_start, section_end): - if lines[index].lstrip().startswith("|"): - header_index = index - break - if header_index is None or header_index + 1 >= section_end: - return None - if not is_markdown_table_separator(lines[header_index + 1]): - return None - - data_start = header_index + 2 - data_end = data_start - while data_end < section_end and lines[data_end].lstrip().startswith("|"): - data_end += 1 - - return PendingLessonsTable( - column_count=len(parse_markdown_table_row(lines[header_index])), - data_start=data_start, - data_end=data_end, - section_end=section_end, - ) - - -def is_markdown_table_separator(line: str) -> bool: - cells = parse_markdown_table_row(line) - return bool(cells) and all(re.fullmatch(r":?-{3,}:?", cell) for cell in cells) - - -def parse_markdown_table_row(line: str) -> list[str]: - stripped = line.strip() - if not stripped.startswith("|"): - return [] - core = stripped.strip("|") - return [cell.strip() for cell in core.split("|")] - - -def normalize_pending_lessons_row(row: list[str], column_count: int) -> list[str]: - if len(row) >= column_count: - return row[:column_count] - return row + [""] * (column_count - len(row)) - - -def format_markdown_table_row(cells: list[str]) -> str: - return f"| {' | '.join(cells)} |" - - -def ensure_trailing_newline(content: str) -> str: - return content if content.endswith("\n") else f"{content}\n" - - -def sha256_text(content: str) -> str: - return hashlib.sha256(content.encode("utf-8")).hexdigest() - - -def discover_source_sync_files(root: Path) -> set[str]: - files = { - relative_path - for relative_path in MANAGED_ROOT_FILES - if (root / relative_path).exists() - } - files.update( - relative_path - for relative_path in MANAGED_WORKFLOW_FILES - if (root / relative_path).exists() - ) - files.update(all_files_under(root, ".github/agents")) - files.update(all_files_under(root, ".github/prompts")) - files.update(all_files_under(root, MANAGED_SKILL_DIR)) - for template_path in CONSUMER_LOCAL_KNOWLEDGE_TEMPLATES.values(): - files.discard(template_path) - return { - relative_path - for relative_path in files - if not is_ignored_sync_path(relative_path) - and not is_consumer_sync_excluded_path(relative_path) - and not is_local_asset(relative_path) - } - - -def discover_target_managed_files(root: Path) -> set[str]: - files = { - relative_path - for relative_path in MANAGED_ROOT_FILES - if (root / relative_path).exists() - } - files.update( - relative_path - for relative_path in MANAGED_WORKFLOW_FILES - if (root / relative_path).exists() - ) - if (root / INVENTORY_PATH).exists(): - files.add(INVENTORY_PATH) - files.update(all_files_under(root, ".github/agents")) - files.update(all_files_under(root, ".github/instructions")) - files.update(all_files_under(root, ".github/prompts")) - files.update(all_files_under(root, MANAGED_SKILL_DIR)) - return { - relative_path - for relative_path in files - if not is_ignored_sync_path(relative_path) - and not is_consumer_sync_excluded_path(relative_path) - } - - -def discover_target_excluded_sync_files(root: Path) -> set[str]: - files: set[str] = set() - files.update(all_files_under(root, ".github/agents")) - files.update(all_files_under(root, ".github/instructions")) - files.update(all_files_under(root, ".github/prompts")) - files.update(all_files_under(root, MANAGED_SKILL_DIR)) - return { - relative_path - for relative_path in files - if not is_ignored_sync_path(relative_path) - and is_consumer_sync_excluded_path(relative_path) - } - - -def detect_target_stacks(root: Path) -> list[str]: - stacks: list[str] = [] - if (root / "pyproject.toml").exists() or any(root.rglob("*.py")): - stacks.append("python") - if ( - (root / "package.json").exists() - or any(root.rglob("*.ts")) - or any(root.rglob("*.js")) - ): - stacks.append("node") - if (root / "go.mod").exists() or any(root.rglob("*.go")): - stacks.append("go") - if any(root.rglob("*.tf")): - stacks.append("terraform") - if any(root.rglob("*.java")) or any(root.rglob("*.kt")): - stacks.append("java") - return stacks or ["unknown"] - - -def render_sync_plan_markdown(plan: SyncPlan) -> str: - lines = [ - "# Copilot Sync Plan", - "", - f"- Source root: `{plan.source_root.as_posix()}`", - f"- Target root: `{plan.target_root.as_posix()}`", - f"- Source revision: `{plan.source_revision or 'unknown'}`", - f"- Source version: `{plan.source_version or 'unknown'}`", - f"- Target manifest source version: `{plan.target_manifest_source_version or 'unknown'}`", - f"- Target dirty: `{'yes' if plan.target_dirty else 'no'}`", - f"- Detected stacks: `{', '.join(plan.stacks)}`", - "", - "## Preserved Local Assets", - "", - ] - if plan.local_assets: - lines.extend(f"- `{path}`" for path in plan.local_assets) - else: - lines.append("No preserved `local-*` assets detected.") - lines.append("") - - action_groups: dict[str, list[SyncOperation]] = {} - for operation in plan.operations: - action_groups.setdefault(operation.action, []).append(operation) - - lines.append("## Planned Operations") - lines.append("") - for action in [ - "create", - "update", - "rename", - "ensure", - "rebuild", - "delete", - "manual", - "preserve", - "unchanged", - ]: - group = action_groups.get(action, []) - if not group: - continue - lines.append(f"### {action.title()}") - lines.append("") - for operation in group: - lines.append(f"- `{operation.path}`: {operation.reason}") - lines.append("") - - return "\n".join(lines).rstrip() + "\n" - - -def write_sync_plan(plan: SyncPlan) -> Path: - plan_path = plan.target_root / SYNC_PLAN_PATH - write_text(plan_path, render_sync_plan_markdown(plan)) - return plan_path - - -def ensure_superpowers_gitignore_entry(current_content: str | None) -> str: - if current_content is None: - return f"{TARGET_SUPERPOWERS_IGNORE_ENTRY}\n" - - lines = current_content.splitlines() - normalized_entries = {line.strip() for line in lines} - accepted_entries = { - "tmp/superpowers", - "tmp/superpowers/", - "/tmp/superpowers", - "/tmp/superpowers/", - } - if normalized_entries & accepted_entries: - return ( - current_content - if current_content.endswith("\n") - else f"{current_content}\n" - ) - - updated = current_content - if updated and not updated.endswith("\n"): - updated += "\n" - updated += f"{TARGET_SUPERPOWERS_IGNORE_ENTRY}\n" - return updated - - -def read_source_version(source_root: Path) -> str | None: - version_path = source_root / VERSION_PATH - if not version_path.exists(): - return None - version = read_text(version_path).strip() - return version or None - - -def read_target_manifest_source_version(target_root: Path) -> str | None: - manifest_path = target_root / SYNC_MANIFEST_PATH - if not manifest_path.exists(): - return None - try: - payload = json.loads(read_text(manifest_path)) - except (OSError, json.JSONDecodeError): - return None - - source_version = payload.get("source_version") - if not isinstance(source_version, str): - return None - normalized = source_version.strip() - return normalized or None - - -def write_sync_manifest(plan: SyncPlan) -> Path: - manifest_path = plan.target_root / SYNC_MANIFEST_PATH - managed_hashes: dict[str, str] = {} - managed_fingerprints: list[dict[str, object]] = [] - managed_settings = { - "/".join(setting_path): value - for setting_path, value in VSCODE_COPILOT_SETTINGS - } - for operation in plan.operations: - if operation.action in {"delete", "manual", "preserve"}: - continue - if operation.path == VSCODE_SETTINGS_PATH: - continue - target_path = plan.target_root / operation.path - if target_path.exists(): - managed_hashes[operation.path] = sha256_file(target_path) - managed_fingerprints.append( - build_fingerprint( - plan.target_root, - target_path, - source_ref_base=plan.source_root.as_posix(), - ).to_dict() - ) - - payload = { - "generated_at": datetime.now(timezone.utc).isoformat(), - "source_root": plan.source_root.as_posix(), - "target_root": plan.target_root.as_posix(), - "source_revision": plan.source_revision, - "source_version": plan.source_version, - "normalization_version": NORMALIZATION_VERSION, - "hash_algo": HASH_ALGO, - "local_assets": list(plan.local_assets), - "managed_settings": managed_settings, - "managed_hashes": managed_hashes, - "managed_fingerprints": managed_fingerprints, - } - write_text(manifest_path, json.dumps(payload, indent=2, sort_keys=True) + "\n") - return manifest_path - - -def apply_sync_plan(plan: SyncPlan, allow_dirty_target: bool = False) -> Path: - manual_operations = [ - operation.path for operation in plan.operations if operation.action == "manual" - ] - if manual_operations: - paths = ", ".join(manual_operations) - raise RuntimeError( - f"Sync plan requires manual reconciliation before apply: {paths}." - ) - - if ( - plan.target_dirty - and not allow_dirty_target - and any( - operation.action - in {"create", "update", "rename", "ensure", "rebuild", "delete"} - for operation in plan.operations - ) - ): - raise RuntimeError( - "Target repository is dirty. Re-run with --allow-dirty-target if this is intentional." - ) - - for operation in plan.operations: - target_path = plan.target_root / operation.path - if operation.action in {"create", "update"}: - target_path.parent.mkdir(parents=True, exist_ok=True) - if operation.path == LESSONS_PATH: - if plan.generated_lessons is None: - raise RuntimeError( - "Generated LESSONS_LEARNED.md content missing from sync plan." - ) - write_text(target_path, plan.generated_lessons) - elif operation.path in CONSUMER_LOCAL_KNOWLEDGE_TEMPLATES: - source_path = ( - plan.source_root - / CONSUMER_LOCAL_KNOWLEDGE_TEMPLATES[operation.path] - ) - copy2(source_path, target_path) - else: - source_path = plan.source_root / operation.path - copy2(source_path, target_path) - elif operation.action == "rename" and operation.path in { - ARCHITECTURE_PATH, - REPOSITORY_CONTEXT_PATH, - }: - legacy_candidates = { - ARCHITECTURE_PATH: ARCHITECTURE_LEGACY_PATHS, - REPOSITORY_CONTEXT_PATH: REPOSITORY_CONTEXT_LEGACY_PATHS, - }[operation.path] - existing_legacy_paths = [ - plan.target_root / legacy_path - for legacy_path in legacy_candidates - if (plan.target_root / legacy_path).exists() - ] - if len(existing_legacy_paths) != 1: - raise RuntimeError( - f"Sync plan requested rename for {operation.path}, but no unique legacy source path is available." - ) - if target_path.exists(): - raise RuntimeError( - f"Sync plan requested rename for {operation.path}, but the canonical target path already exists." - ) - legacy_path = existing_legacy_paths[0] - target_path.parent.mkdir(parents=True, exist_ok=True) - legacy_path.rename(target_path) - cleanup_empty_parents(legacy_path, plan.target_root) - elif operation.action == "delete": - if target_path.exists(): - target_path.unlink() - cleanup_empty_parents(target_path, plan.target_root) - elif operation.action == "ensure" and operation.path == TARGET_GITIGNORE_PATH: - if plan.generated_gitignore is None: - raise RuntimeError( - "Generated .gitignore content missing from sync plan." - ) - write_text(target_path, plan.generated_gitignore) - elif operation.action == "ensure" and operation.path == VSCODE_SETTINGS_PATH: - if target_path.exists(): - current = read_text(target_path) - updated = apply_managed_vscode_copilot_settings(current) - else: - updated = render_minimal_vscode_settings_jsonc() - write_text(target_path, updated) - elif operation.action == "rebuild" and operation.path == INVENTORY_PATH: - write_text(target_path, plan.generated_inventory) - - manifest_path = write_sync_manifest(plan) - clear_sync_plan(plan) - return manifest_path - - -def clear_sync_plan(plan: SyncPlan) -> None: - plan_path = plan.target_root / SYNC_PLAN_PATH - if not plan_path.exists(): - return - plan_path.unlink() - cleanup_empty_parents(plan_path, plan.target_root) - - -def cleanup_empty_parents(path: Path, stop_at: Path) -> None: - current = path.parent - while current != stop_at and current.exists(): - if any(current.iterdir()): - return - current.rmdir() - current = current.parent diff --git a/.github/scripts/run.sh b/.github/scripts/run.sh index a7a126e4..1c122019 100755 --- a/.github/scripts/run.sh +++ b/.github/scripts/run.sh @@ -3,7 +3,6 @@ # Purpose: Bootstrap the local Python environment and run a Copilot maintenance tool. # Usage examples: # ./.github/scripts/run.sh build_inventory --root . -# ./.github/scripts/run.sh sync_copilot_catalog plan --target-repo ../consumer-repo set -Eeuo pipefail @@ -40,7 +39,6 @@ Tools: audit_copilot_catalog detect_token_risks sync_home_ai_resources - sync_copilot_catalog validate_critical_output validate_internal_skills EOF @@ -140,9 +138,6 @@ resolve_script() { sync_home_ai_resources|sync_home_ai_resources.py) printf '%s\n' "$REPO_ROOT/.github/skills/local-agent-sync-install-ai-resources/scripts/run.sh" ;; - sync_copilot_catalog|sync_copilot_catalog.py) - printf '%s\n' "$SCRIPT_DIR/sync_copilot_catalog.py" - ;; validate_critical_output) printf '%s\n' "$REPO_ROOT/.github/skills/internal-gateway-critical-master/scripts/validate_critical_output.py" ;; diff --git a/.github/scripts/sync_copilot_catalog.py b/.github/scripts/sync_copilot_catalog.py deleted file mode 100644 index baa2d5ae..00000000 --- a/.github/scripts/sync_copilot_catalog.py +++ /dev/null @@ -1,229 +0,0 @@ -#!/usr/bin/env python3 -"""Purpose: plan and apply source-authoritative Copilot catalog sync operations. - -Usage examples: - python3 ./.github/scripts/sync_copilot_catalog.py plan --target-repo ../consumer-repo - python3 ./.github/scripts/sync_copilot_catalog.py apply --target-repo ../consumer-repo --allow-dirty-target - -Generated files: - - Plan output: tmp/copilot-sync.plan.md - - Canonical sync manifest on apply: .github/copilot-sync.manifest.json -""" - -from __future__ import annotations - -import argparse -from collections import Counter -from pathlib import Path - -from lib.catalog_checks import run_consistency_checks -from lib.fingerprinting import HASH_ALGO, NORMALIZATION_VERSION -from lib.shared import ( - Finding, - find_repo_root, - log_error, - log_info, - log_success, - render_json, -) -from lib.syncing import apply_sync_plan, build_sync_plan, write_sync_plan - - -def parse_args() -> argparse.Namespace: - parser = argparse.ArgumentParser(description="Plan or apply Copilot catalog sync operations.") - parser.add_argument("command", choices=["plan", "apply"], help="Run a dry plan or apply the planned changes.") - parser.add_argument("--source-root", default=".", help="Source standards repository root.") - parser.add_argument("--target-repo", required=True, help="Target repository root.") - parser.add_argument("--allow-dirty-target", action="store_true", help="Allow apply against a dirty target worktree.") - parser.add_argument("--format", choices=["text", "json", "compact"], default="text", help="Output format.") - return parser.parse_args() - - -def main() -> int: - args = parse_args() - source_root = find_repo_root(Path(args.source_root)) - target_root = find_repo_root(Path(args.target_repo)) - source_findings = run_consistency_checks(source_root, include_token_risks=True) - blocking_source_findings = [finding for finding in source_findings if finding.severity == "blocking"] - - plan = build_sync_plan(source_root, target_root) - plan_path = write_sync_plan(plan) - - if args.command == "apply" and blocking_source_findings: - if args.format == "compact": - print( - render_json( - build_compact_payload( - mode="apply", - plan=plan, - plan_path=plan_path, - source_findings=source_findings, - manifest_path=None, - status="blocked", - ) - ) - ) - render_source_findings(blocking_source_findings) - log_error("Source repository has blocking governance findings; sync apply aborted.") - return 1 - - if args.command == "apply": - try: - manifest_path = apply_sync_plan(plan, allow_dirty_target=args.allow_dirty_target) - except RuntimeError as error: - log_error(str(error)) - return 1 - if args.format == "json": - print( - render_json( - { - "mode": "apply", - "plan": plan.to_dict(), - "plan_path": plan_path.as_posix(), - "manifest_path": manifest_path.as_posix(), - "normalization_version": NORMALIZATION_VERSION, - "hash_algo": HASH_ALGO, - } - ) - ) - elif args.format == "compact": - print( - render_json( - build_compact_payload( - mode="apply", - plan=plan, - plan_path=plan_path, - source_findings=source_findings, - manifest_path=manifest_path, - status="ok", - ) - ) - ) - else: - render_text("apply", plan, plan_path, manifest_path) - log_success("Sync apply completed.") - return 0 - - if args.format == "json": - print( - render_json( - { - "mode": "plan", - "plan": plan.to_dict(), - "plan_path": plan_path.as_posix(), - "source_findings": [finding.to_dict() for finding in source_findings], - "normalization_version": NORMALIZATION_VERSION, - "hash_algo": HASH_ALGO, - } - ) - ) - elif args.format == "compact": - print( - render_json( - build_compact_payload( - mode="plan", - plan=plan, - plan_path=plan_path, - source_findings=source_findings, - manifest_path=None, - status="ok", - ) - ) - ) - else: - render_text("plan", plan, plan_path) - render_source_findings(source_findings) - return 0 - - -def build_compact_payload( - *, - mode: str, - plan, - plan_path: Path, - source_findings: list[Finding], - manifest_path: Path | None, - status: str, -) -> dict[str, object]: - operation_counts = Counter(operation.action for operation in plan.operations) - severity_counts = Counter(finding.severity for finding in source_findings) - operation_sample = [ - { - "action": operation.action, - "path": operation.path, - "reason": operation.reason, - } - for operation in plan.operations[:10] - ] - - if mode == "plan": - next_action = { - "action": "apply", - "allowed": status == "ok", - "requires_explicit_approval": True, - "reason": ( - "Blocking source findings must be resolved before apply." - if status != "ok" - else "Plan is ready for explicit apply approval." - ), - } - else: - next_action = { - "action": "done", - "allowed": status == "ok", - "requires_explicit_approval": False, - "reason": ( - "Apply aborted due to blocking source findings." - if status != "ok" - else "Apply completed successfully." - ), - } - - payload: dict[str, object] = { - "mode": mode, - "status": status, - "target_repo": plan.target_root.as_posix(), - "plan_path": plan_path.as_posix(), - "operation_counts": { - "total": len(plan.operations), - "by_action": dict(sorted(operation_counts.items())), - }, - "operation_sample": operation_sample, - "source_findings": { - "total": len(source_findings), - "blocking": severity_counts.get("blocking", 0), - "notice": severity_counts.get("notice", 0), - }, - "next_action": next_action, - } - if manifest_path is not None: - payload["manifest_path"] = manifest_path.as_posix() - return payload - - -def render_text(mode: str, plan, plan_path: Path, manifest_path: Path | None = None) -> None: - if mode == "apply": - log_info(f"Sync apply completed for {plan.target_root.as_posix()}.") - else: - log_info(f"Sync {mode} ready for {plan.target_root.as_posix()}.") - if plan_path.exists(): - log_info(f"Plan file: {plan_path.as_posix()}") - else: - log_info(f"Plan file cleared: {plan_path.as_posix()}") - if manifest_path is not None: - log_info(f"Manifest file: {manifest_path.as_posix()}") - log_info(f"Fingerprinting: {HASH_ALGO} normalized-content ({NORMALIZATION_VERSION})") - for operation in plan.operations: - print(f"- {operation.action:9s} {operation.path} :: {operation.reason}") - - -def render_source_findings(findings: list[Finding]) -> None: - if not findings: - return - log_info("Source audit findings:") - for finding in findings: - print(f"- {finding.severity} :: {finding.path} :: {finding.code} :: {finding.message}") - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/.github/skills/addyosmani-code-review-and-quality/SKILL.md b/.github/skills/addyosmani-code-review-and-quality/SKILL.md index ad789bf8..3e58e23e 100644 --- a/.github/skills/addyosmani-code-review-and-quality/SKILL.md +++ b/.github/skills/addyosmani-code-review-and-quality/SKILL.md @@ -289,6 +289,16 @@ Part of code review is dependency review: **Rule:** Prefer standard library and existing utilities over new dependencies. Every dependency is a liability. +**Upgrading an existing dependency** is a code change like any other, and the riskiest upgrades are the ones merged in bulk with a message like "bump deps." Review them with the same discipline: + +1. **Read the changelog, not just the version number.** Semver is a promise the maintainer may not have kept — a "patch" can carry a behavioral change. For a major bump, read the migration notes and find what breaks. +2. **One dependency per change.** Upgrade and merge them individually (or in small related groups). When a bulk bump breaks the build, you've lost which package did it; a single-package change makes the cause obvious and the revert clean. +3. **Let the tests decide.** The upgrade is verified by a green suite before *and* after, not by "it installed." If coverage around the dependency's behavior is thin, that gap is the real finding — add a test first. +4. **Mind the transitive graph.** Most installed packages are ones nobody chose directly. Review the lockfile diff, not just `package.json`; a single direct bump can pull in dozens of indirect changes. +5. **Keep the lockfile honest.** Commit it, review its diff, and never hand-edit it. The lockfile is the thing that actually pins what ships. + +For triaging `npm audit` findings and supply-chain risk (typosquatting, compromised maintainers), follow the `security-and-hardening` skill — this section covers the upgrade *workflow*, that one covers the security verdict. + ## The Review Checklist ```markdown @@ -352,6 +362,8 @@ Part of code review is dependency review: | "The tests pass, so it's good" | Tests are necessary but not sufficient. They don't catch architecture problems, security issues, or readability concerns. | | "The refactor makes it cleaner" | Relocating complexity isn't reducing it. If the reader still holds the same number of concepts, the structure didn't improve — look for the version where branches disappear. | | "It's only a small addition to this file" | Small diffs still push files past a healthy size and bolt branches onto unrelated flows. Judge the resulting structure, not the diff size. | +| "It's just a version bump" | A bump is a behavior change you didn't write. Read the changelog; semver doesn't guarantee no breakage. | +| "I'll upgrade everything in one PR to save time" | A bulk bump that breaks the build hides which package did it. One dependency per change keeps the cause and the revert clean. | ## Red Flags @@ -367,6 +379,8 @@ Part of code review is dependency review: - A change that grows an already-large file instead of decomposing it - New conditionals scattered into unrelated code paths (a missing abstraction) - A bespoke helper that duplicates an existing canonical one, or feature logic placed in a shared module +- A bulk "bump dependencies" PR with no changelog review and no per-package isolation +- A lockfile change that's hand-edited, uncommitted, or merged without reviewing its diff ## Verification @@ -377,5 +391,6 @@ After review is complete: - [ ] Tests pass - [ ] Build succeeds - [ ] The verification story is documented (what changed, how it was verified) +- [ ] Dependency upgrades were reviewed against their changelog, isolated per package, and verified by a green suite with the lockfile diff reviewed **Presumptive blockers:** surface and propose the simpler design for each of these; escalate to Required only when the change actively makes structure worse: a refactor that relocates complexity instead of reducing it; a change that pushes a file past the size boundary with no decomposition; feature logic added to a shared module; a near-duplicate of an existing canonical helper; a silent fallback that hides an unclear invariant. diff --git a/.github/skills/anthropic-docx/LICENSE.txt b/.github/skills/anthropic-docx/LICENSE.txt new file mode 100644 index 00000000..c55ab422 --- /dev/null +++ b/.github/skills/anthropic-docx/LICENSE.txt @@ -0,0 +1,30 @@ +© 2025 Anthropic, PBC. All rights reserved. + +LICENSE: Use of these materials (including all code, prompts, assets, files, +and other components of this Skill) is governed by your agreement with +Anthropic regarding use of Anthropic's services. If no separate agreement +exists, use is governed by Anthropic's Consumer Terms of Service or +Commercial Terms of Service, as applicable: +https://www.anthropic.com/legal/consumer-terms +https://www.anthropic.com/legal/commercial-terms +Your applicable agreement is referred to as the "Agreement." "Services" are +as defined in the Agreement. + +ADDITIONAL RESTRICTIONS: Notwithstanding anything in the Agreement to the +contrary, users may not: + +- Extract these materials from the Services or retain copies of these + materials outside the Services +- Reproduce or copy these materials, except for temporary copies created + automatically during authorized use of the Services +- Create derivative works based on these materials +- Distribute, sublicense, or transfer these materials to any third party +- Make, offer to sell, sell, or import any inventions embodied in these + materials +- Reverse engineer, decompile, or disassemble these materials + +The receipt, viewing, or possession of these materials does not convey or +imply any license or right beyond those expressly granted above. + +Anthropic retains all right, title, and interest in these materials, +including all copyrights, patents, and other intellectual property rights. diff --git a/.github/skills/anthropic-docx/SKILL.md b/.github/skills/anthropic-docx/SKILL.md new file mode 100644 index 00000000..2b1fe9c2 --- /dev/null +++ b/.github/skills/anthropic-docx/SKILL.md @@ -0,0 +1,91 @@ +--- +name: anthropic-docx +description: "Use this skill whenever the user wants to create, read, edit, or manipulate Word documents (.docx files) or Word templates (.dotx files). Triggers include: any mention of 'Word doc', 'word document', '.docx', '.dotx', or requests to produce professional documents with formatting like tables of contents, headings, page numbers, or letterheads. Also use when extracting or reorganizing content from .docx or .dotx files, inserting or replacing images in documents, performing find-and-replace in Word files, working with tracked changes or comments, or converting content into a polished Word document. If the user asks for a 'report', 'memo', 'letter', 'template', or similar deliverable as a Word or .docx file, use this skill. Do NOT use for PDFs, spreadsheets, Google Docs, or general coding tasks unrelated to document generation." +license: Proprietary. LICENSE.txt has complete terms +--- + +# DOCX creation, editing, and analysis + +A `.docx` is a ZIP archive of XML files. Choose your approach by task: + +| Task | Approach | +|---|---| +| **Create** a new document | Write a `anthropic-docx` (npm) script — see gotchas below | +| **Edit** an existing document | `unzip` → edit `word/document.xml` → `zip` (docx-js cannot open existing files) | +| **Read** content | `pandoc -t markdown file.docx` | + +> Script paths below are relative to this skill's directory. + +## Creating with docx-js — gotchas + +`anthropic-docx` is preinstalled — do not run `npm install` first; write the script and `require('docx')` directly. Only if that require fails: `npm install docx`. The model knows the API; these are the footguns: + +- **Page size defaults to A4.** For US Letter set `page: { size: { width: 12240, height: 15840 } }` (DXA; 1440 = 1″). +- **Landscape:** pass portrait dimensions and `orientation: PageOrientation.LANDSCAPE` — docx-js swaps width/height internally. +- **Tables need dual widths:** set `columnWidths` on the table AND `width` on every cell, both in `WidthType.DXA` (PERCENTAGE breaks in Google Docs). Column widths must sum to the table width. +- **Table shading:** use `ShadingType.CLEAR`, never `SOLID` (renders black). +- **Lists:** never insert `•` literally; use a `numbering` config with `LevelFormat.BULLET`. +- **`ImageRun` requires `type:`** (`"png"`, `"jpg"`, …). +- **`PageBreak` must be inside a `Paragraph`.** +- **Never use `\n`** — use separate `Paragraph` elements. +- **TOC:** headings must use built-in `HeadingLevel.*`; custom heading styles need `outlineLevel` set or they won't appear. +- **Don't use a table as a horizontal rule** — use a paragraph bottom border instead. +- **Dot-leader / right-aligned-on-same-line:** use `PositionalTab` (`alignment: PositionalTabAlignment.RIGHT`, `leader: PositionalTabLeader.DOT`) inside a `TextRun`, not literal `.` or space padding. + +## Verify the output + +After writing a `.docx`, render it and look at it: + +```bash +python scripts/office/soffice.py --headless --convert-to pdf output.docx +pdftoppm -jpeg -r 100 output.pdf page +ls page-*.jpg # then Read the images +``` + +`pdftoppm` zero-pads page numbers to the width of the page count (`page-01.jpg`…`page-12.jpg`). + +## Editing existing documents + +Legacy `.doc` files must be converted first: `python scripts/office/soffice.py --headless --convert-to docx file.doc`. + +```bash +unzip -q doc.docx -d unpacked/ +find unpacked -type l -delete # strip symlink entries — docx from external parties is untrusted +python scripts/merge_runs.py unpacked/ # coalesce fragmented runs so text is findable +# edit unpacked/word/document.xml in place — do NOT reformat or pretty-print +(cd unpacked && rm -f ../out.docx && zip -Xr ../out.docx .) +python scripts/office/validate.py out.docx --original doc.docx # XSD checks; --auto-repair fixes common issues +# redlining? add --author "" to check every edit is tracked +``` + +Word splits text across many `` runs (revision ids, spell-check markers), so a phrase you can see in the document often doesn't exist as a contiguous string in the XML. `merge_runs.py` merges adjacent identically-formatted runs in `word/document.xml` without changing content or rendering; it also accepts a `.docx` directly (`python scripts/merge_runs.py doc.docx -o merged.docx`). + +**Tracked changes:** when redlining, validate with `--author ""` (needs `--original`) — it reports any text you changed without a ``/`` around it, which is easy to do by accident and invisible in the accepted view. Wrap runs in ``/`` with `w:id`, `w:author`, `w:date` attributes. Inside ``, the text element is ``, not ``. A deleted paragraph mark (``) means "merge this paragraph into the next" — so deleting a paragraph outright is that plus a `` around every run. The `` must come before the rPr's other children; their order is schema-enforced. + +To produce a clean copy with all tracked changes accepted: `python scripts/accept_changes.py in.docx out.docx`. + +Accepting a deleted paragraph mark should join that paragraph to the one below it, so a paragraph whose runs are *all* deleted vanishes. Word does this; `accept_changes.py` and `pandoc --track-changes=accept` don't always. Both fail the same way — they strip the deleted text but leave the emptied paragraph behind, which reads as a stray empty bullet when it was auto-numbered: + +- `pandoc --track-changes=accept` never joins the paragraphs. +- `accept_changes.py` (LibreOffice) joins them correctly, except when the deleted paragraph is followed by an empty spacer paragraph. + +An empty bullet in either view is an artifact of that view, not a defect in the document. Check paragraph deletions in the XML. + +## Comments + +Comments require six cross-linked files. Use the helper — directory mode when you'll also be editing `document.xml` (saves an unzip/rezip cycle), `.docx`-direct mode otherwise: + +```bash +# Against an already-unpacked directory (preferred when also placing markers) +python scripts/comment.py unpacked/ "Fees & expenses cap is too low" +python scripts/comment.py unpacked/ "Agreed" --parent 0 + +# Against a .docx directly +python scripts/comment.py contract.docx "This cap is too low" -o annotated.docx +``` + +The script writes `comments.xml`, `commentsExtended.xml`, `commentsIds.xml`, `commentsExtensible.xml`, the relationships, and the content-type overrides. Comment IDs are auto-assigned. It then prints the ``/``/`` snippet to add to `word/document.xml` so the comment anchors to specific text — until you place those markers, the comment exists but is not visible. + +## Dependencies + +`anthropic-docx` (npm, preinstalled — install only if `require('docx')` fails) · `pandoc` · LibreOffice (`soffice`) · `pdftoppm` (Poppler) diff --git a/.github/skills/anthropic-docx/scripts/__init__.py b/.github/skills/anthropic-docx/scripts/__init__.py new file mode 100644 index 00000000..8b137891 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/__init__.py @@ -0,0 +1 @@ + diff --git a/.github/skills/anthropic-docx/scripts/accept_changes.py b/.github/skills/anthropic-docx/scripts/accept_changes.py new file mode 100644 index 00000000..8e363161 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/accept_changes.py @@ -0,0 +1,135 @@ +"""Accept all tracked changes in a DOCX file using LibreOffice. + +Requires LibreOffice (soffice) to be installed. +""" + +import argparse +import logging +import shutil +import subprocess +from pathlib import Path + +from office.soffice import get_soffice_env + +logger = logging.getLogger(__name__) + +LIBREOFFICE_PROFILE = "/tmp/libreoffice_docx_profile" +MACRO_DIR = f"{LIBREOFFICE_PROFILE}/user/basic/Standard" + +ACCEPT_CHANGES_MACRO = """ + + + Sub AcceptAllTrackedChanges() + Dim document As Object + Dim dispatcher As Object + + document = ThisComponent.CurrentController.Frame + dispatcher = createUnoService("com.sun.star.frame.DispatchHelper") + + dispatcher.executeDispatch(document, ".uno:AcceptAllTrackedChanges", "", 0, Array()) + ThisComponent.store() + ThisComponent.close(True) + End Sub +""" + + +def accept_changes( + input_file: str, + output_file: str, +) -> tuple[None, str]: + input_path = Path(input_file) + output_path = Path(output_file) + + if not input_path.exists(): + return None, f"Error: Input file not found: {input_file}" + + if not input_path.suffix.lower() == ".docx": + return None, f"Error: Input file is not a DOCX file: {input_file}" + + try: + output_path.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(input_path, output_path) + except Exception as e: + return None, f"Error: Failed to copy input file to output location: {e}" + + if not _setup_libreoffice_macro(): + return None, "Error: Failed to setup LibreOffice macro" + + cmd = [ + "soffice", + "--headless", + f"-env:UserInstallation=file://{LIBREOFFICE_PROFILE}", + "--norestore", + "vnd.sun.star.script:Standard.Module1.AcceptAllTrackedChanges?language=Basic&location=application", + str(output_path.absolute()), + ] + + try: + result = subprocess.run( + cmd, + capture_output=True, + text=True, + timeout=30, + check=False, + env=get_soffice_env(), + ) + except subprocess.TimeoutExpired: + return ( + None, + f"Successfully accepted all tracked changes: {input_file} -> {output_file}", + ) + + if result.returncode != 0: + return None, f"Error: LibreOffice failed: {result.stderr}" + + return ( + None, + f"Successfully accepted all tracked changes: {input_file} -> {output_file}", + ) + + +def _setup_libreoffice_macro() -> bool: + macro_dir = Path(MACRO_DIR) + macro_file = macro_dir / "Module1.xba" + + if macro_file.exists() and "AcceptAllTrackedChanges" in macro_file.read_text(): + return True + + if not macro_dir.exists(): + subprocess.run( + [ + "soffice", + "--headless", + f"-env:UserInstallation=file://{LIBREOFFICE_PROFILE}", + "--terminate_after_init", + ], + capture_output=True, + timeout=10, + check=False, + env=get_soffice_env(), + ) + macro_dir.mkdir(parents=True, exist_ok=True) + + try: + macro_file.write_text(ACCEPT_CHANGES_MACRO) + return True + except Exception as e: + logger.warning(f"Failed to setup LibreOffice macro: {e}") + return False + + +if __name__ == "__main__": + parser = argparse.ArgumentParser( + description="Accept all tracked changes in a DOCX file" + ) + parser.add_argument("input_file", help="Input DOCX file with tracked changes") + parser.add_argument( + "output_file", help="Output DOCX file (clean, no tracked changes)" + ) + args = parser.parse_args() + + _, message = accept_changes(args.input_file, args.output_file) + print(message) + + if "Error" in message: + raise SystemExit(1) diff --git a/.github/skills/anthropic-docx/scripts/comment.py b/.github/skills/anthropic-docx/scripts/comment.py new file mode 100644 index 00000000..46ed5f52 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/comment.py @@ -0,0 +1,368 @@ +"""Add comments to a DOCX document. + +Accepts either an unpacked directory OR a .docx/.dotx file directly. + +Usage: + # Against an unpacked directory (writes satellite files in place) + python comment.py unpacked/ "Comment text" + python comment.py unpacked/ "Reply text" --parent 0 + + # Against a .docx directly (extracts, writes satellite files, rezips) + python comment.py contract.docx "This cap is too low" -o annotated.docx + python comment.py contract.docx "Comment" --id 5 # explicit ID + +The comment ID is auto-assigned (max existing + 1) unless --id is given. +Plain text is XML-escaped automatically; if you pass already-escaped text +(e.g. &, ’) use --raw to skip escaping. + +After running, add markers to word/document.xml so the comment is visible: + + ... commented content ... + + +""" + +import argparse +import random +import shutil +import sys +import tempfile +import zipfile +from datetime import datetime, timezone +from pathlib import Path + +import defusedxml.minidom +from xml.parsers.expat import ExpatError +from xml.sax.saxutils import escape as xml_escape + +from office.helpers import opc_target, rezip as _rezip, safe_extract as _safe_extract + +TEMPLATE_DIR = Path(__file__).parent / "templates" +NS = { + "w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main", + "w14": "http://schemas.microsoft.com/office/word/2010/wordml", + "w15": "http://schemas.microsoft.com/office/word/2012/wordml", + "w16cid": "http://schemas.microsoft.com/office/word/2016/wordml/cid", + "w16cex": "http://schemas.microsoft.com/office/word/2018/wordml/cex", +} + +COMMENT_XML = """\ + + + + + + + + + + + + + {text} + + +""" + +COMMENT_MARKER_TEMPLATE = """ +Add to word/document.xml (markers must be direct children of w:p, never inside w:r): + + ... + + """ + +REPLY_MARKER_TEMPLATE = """ +Nest markers inside parent {pid}'s markers (direct children of w:p, never inside w:r): + + ... + + + """ + +SMART_QUOTE_ENTITIES = { + "“": "“", + "”": "”", + "‘": "‘", + "’": "’", +} + + +def _generate_hex_id() -> str: + return f"{random.randint(0, 0x7FFFFFFE):08X}" + + +def _encode_smart_quotes(text: str) -> str: + for char, entity in SMART_QUOTE_ENTITIES.items(): + text = text.replace(char, entity) + return text + + +def _append_xml(xml_path: Path, root_tag: str, content: str) -> None: + dom = defusedxml.minidom.parseString(xml_path.read_text(encoding="utf-8")) + root = dom.getElementsByTagName(root_tag)[0] + ns_attrs = " ".join(f'xmlns:{k}="{v}"' for k, v in NS.items()) + wrapper_dom = defusedxml.minidom.parseString(f"{content}") + for child in wrapper_dom.documentElement.childNodes: + if child.nodeType == child.ELEMENT_NODE: + root.appendChild(dom.importNode(child, True)) + output = _encode_smart_quotes(dom.toxml(encoding="UTF-8").decode("utf-8")) + xml_path.write_text(output, encoding="utf-8") + + +def _find_para_id(comments_path: Path, comment_id: int) -> str | None: + dom = defusedxml.minidom.parseString(comments_path.read_text(encoding="utf-8")) + for c in dom.getElementsByTagName("w:comment"): + if c.getAttribute("w:id") == str(comment_id): + for p in c.getElementsByTagName("w:p"): + if pid := p.getAttribute("w14:paraId"): + return pid + return None + + +def _next_comment_id(comments_path: Path) -> int: + if not comments_path.exists(): + return 0 + dom = defusedxml.minidom.parseString(comments_path.read_text(encoding="utf-8")) + ids = [] + for c in dom.getElementsByTagName("w:comment"): + try: + ids.append(int(c.getAttribute("w:id"))) + except ValueError: + pass + return (max(ids) + 1) if ids else 0 + + +def _get_next_rid(rels_path: Path) -> int: + dom = defusedxml.minidom.parseString(rels_path.read_text(encoding="utf-8")) + max_rid = 0 + for rel in dom.getElementsByTagName("Relationship"): + rid = rel.getAttribute("Id") + if rid and rid.startswith("rId"): + try: + max_rid = max(max_rid, int(rid[3:])) + except ValueError: + pass + return max_rid + 1 + + +def _has_relationship(rels_path: Path, target: str) -> bool: + dom = defusedxml.minidom.parseString(rels_path.read_text(encoding="utf-8")) + return any( + rel.getAttribute("Target") == target + for rel in dom.getElementsByTagName("Relationship") + ) + + +def _has_content_type(ct_path: Path, part_name: str) -> bool: + dom = defusedxml.minidom.parseString(ct_path.read_text(encoding="utf-8")) + return any( + o.getAttribute("PartName") == part_name + for o in dom.getElementsByTagName("Override") + ) + + +_COMMENT_RELS = [ + ("http://schemas.openxmlformats.org/officeDocument/2006/relationships/comments", "comments.xml"), + ("http://schemas.microsoft.com/office/2011/relationships/commentsExtended", "commentsExtended.xml"), + ("http://schemas.microsoft.com/office/2016/09/relationships/commentsIds", "commentsIds.xml"), + ("http://schemas.microsoft.com/office/2018/08/relationships/commentsExtensible", "commentsExtensible.xml"), +] +_COMMENT_OVERRIDES = [ + ("/word/comments.xml", "application/vnd.openxmlformats-officedocument.wordprocessingml.comments+xml"), + ("/word/commentsExtended.xml", "application/vnd.openxmlformats-officedocument.wordprocessingml.commentsExtended+xml"), + ("/word/commentsIds.xml", "application/vnd.openxmlformats-officedocument.wordprocessingml.commentsIds+xml"), + ("/word/commentsExtensible.xml", "application/vnd.openxmlformats-officedocument.wordprocessingml.commentsExtensible+xml"), +] + + +def _ensure_comment_relationships(unpacked_dir: Path) -> None: + rels_path = unpacked_dir / "word" / "_rels" / "document.xml.rels" + if not rels_path.exists(): + return + dom = defusedxml.minidom.parseString(rels_path.read_text(encoding="utf-8")) + root = dom.documentElement + comment_types = {rel_type for rel_type, _ in _COMMENT_RELS} + existing = set() + for rel in dom.getElementsByTagName("Relationship"): + if rel.getAttribute("Type") not in comment_types: + continue + part = opc_target( + rel.getAttribute("Target"), + "word/document.xml", + rel.getAttribute("TargetMode"), + ) + if part is not None: + existing.add(part) + next_rid = _get_next_rid(rels_path) + changed = False + for rel_type, target in _COMMENT_RELS: + if opc_target(target, "word/document.xml") in existing: + continue + rel = dom.createElement("Relationship") + rel.setAttribute("Id", f"rId{next_rid}") + rel.setAttribute("Type", rel_type) + rel.setAttribute("Target", target) + root.appendChild(rel) + next_rid += 1 + changed = True + if changed: + rels_path.write_bytes(dom.toxml(encoding="UTF-8")) + + +def _ensure_comment_content_types(unpacked_dir: Path) -> None: + ct_path = unpacked_dir / "[Content_Types].xml" + if not ct_path.exists(): + return + dom = defusedxml.minidom.parseString(ct_path.read_text(encoding="utf-8")) + root = dom.documentElement + existing = { + o.getAttribute("PartName") + for o in dom.getElementsByTagName("Override") + } + changed = False + for part_name, content_type in _COMMENT_OVERRIDES: + if part_name in existing: + continue + override = dom.createElement("Override") + override.setAttribute("PartName", part_name) + override.setAttribute("ContentType", content_type) + root.appendChild(override) + changed = True + if changed: + ct_path.write_bytes(dom.toxml(encoding="UTF-8")) + + +def add_comment( + unpacked_dir: Path | str, + text: str, + comment_id: int | None = None, + author: str = "Claude", + initials: str = "C", + parent_id: int | None = None, + raw: bool = False, +) -> tuple[int, str, str]: + unpacked_dir = Path(unpacked_dir) + if not raw: + text = xml_escape(text) + author = xml_escape(author, {'"': """}) + initials = xml_escape(initials, {'"': """}) + word = unpacked_dir / "word" + if not word.exists(): + raise FileNotFoundError(f"{word} not found (not an unpacked .docx?)") + + comments = word / "comments.xml" + if comment_id is None: + comment_id = _next_comment_id(comments) + + parent_para = None + if parent_id is not None: + parent_para = _find_para_id(comments, parent_id) if comments.exists() else None + if not parent_para: + raise ValueError(f"parent comment {parent_id} not found") + + para_id, durable_id = _generate_hex_id(), _generate_hex_id() + ts = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + + if not comments.exists(): + shutil.copy(TEMPLATE_DIR / "comments.xml", comments) + _ensure_comment_relationships(unpacked_dir) + _ensure_comment_content_types(unpacked_dir) + _append_xml( + comments, + "w:comments", + COMMENT_XML.format( + id=comment_id, author=author, date=ts, initials=initials, + para_id=para_id, text=text, + ), + ) + + ext = word / "commentsExtended.xml" + if not ext.exists(): + shutil.copy(TEMPLATE_DIR / "commentsExtended.xml", ext) + if parent_para is not None: + _append_xml( + ext, "w15:commentsEx", + f'', + ) + else: + _append_xml( + ext, "w15:commentsEx", + f'', + ) + + ids = word / "commentsIds.xml" + if not ids.exists(): + shutil.copy(TEMPLATE_DIR / "commentsIds.xml", ids) + _append_xml( + ids, "w16cid:commentsIds", + f'', + ) + + extensible = word / "commentsExtensible.xml" + if not extensible.exists(): + shutil.copy(TEMPLATE_DIR / "commentsExtensible.xml", extensible) + _append_xml( + extensible, "w16cex:commentsExtensible", + f'', + ) + + action = "reply" if parent_id is not None else "comment" + return comment_id, para_id, f"Added {action} id={comment_id} (paraId={para_id})" + + +def main() -> None: + p = argparse.ArgumentParser(description="Add a comment to a DOCX (directory or .docx file).") + p.add_argument("input", help="Unpacked DOCX directory OR a .docx/.dotx file") + p.add_argument("text", help="Comment text (plain text; XML-escaped automatically)") + p.add_argument("--raw", action="store_true", + help="Treat text as pre-escaped XML (skip automatic escaping)") + p.add_argument("--id", type=int, dest="comment_id", + help="Comment ID (default: auto-assign as max existing + 1)") + p.add_argument("--author", default="Claude", help="Author name") + p.add_argument("--initials", default="C", help="Author initials") + p.add_argument("--parent", type=int, help="Parent comment ID (makes this a reply)") + p.add_argument("-o", "--output", + help="Output .docx path (only used when input is a .docx; default: overwrite input)") + args = p.parse_args() + + src = Path(args.input) + + try: + if src.is_dir(): + if args.output: + print("Warning: --output ignored for directory input", file=sys.stderr) + cid, _, msg = add_comment( + src, args.text, comment_id=args.comment_id, + author=args.author, initials=args.initials, + parent_id=args.parent, raw=args.raw, + ) + print(msg) + elif src.is_file() and src.suffix.lower() in (".docx", ".dotx"): + out = Path(args.output) if args.output else src + with tempfile.TemporaryDirectory() as tmp: + tmp_path = Path(tmp) + with zipfile.ZipFile(src) as zf: + _safe_extract(zf, tmp_path) + cid, _, msg = add_comment( + tmp_path, args.text, comment_id=args.comment_id, + author=args.author, initials=args.initials, + parent_id=args.parent, raw=args.raw, + ) + _rezip(tmp_path, out) + print(msg) + print(f"Wrote {out} (comment defined; add markers to word/document.xml to make it visible)") + else: + print(f"Error: {src} is neither a directory nor a .docx/.dotx file", file=sys.stderr) + sys.exit(1) + except (FileNotFoundError, ValueError, zipfile.BadZipFile, ExpatError) as e: + print(f"Error: {e}", file=sys.stderr) + sys.exit(1) + + if args.parent is not None: + print(REPLY_MARKER_TEMPLATE.format(pid=args.parent, cid=cid)) + else: + print(COMMENT_MARKER_TEMPLATE.format(cid=cid)) + + +if __name__ == "__main__": + main() diff --git a/.github/skills/anthropic-docx/scripts/merge_runs.py b/.github/skills/anthropic-docx/scripts/merge_runs.py new file mode 100644 index 00000000..4c7c1bf2 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/merge_runs.py @@ -0,0 +1,310 @@ +"""Merge adjacent identically-formatted runs in a DOCX. + +Word fragments paragraph text across many elements (revision ids, +spell-check markers, editing history), which makes find-and-replace on +word/document.xml unreliable — the string you're looking for is split +across runs. This coalesces adjacent runs whose formatting () is +identical, strips rsid attributes and proofErr markers, and consolidates the +text elements — , and for text inside a tracked deletion. + +Rendering is unchanged. The text you search is what Word draws, which is not +always the bytes in the file: an element without xml:space="preserve" has its +edge whitespace trimmed before it reaches the page, so `Hello ` +followed by `world` reads "Helloworld" and merges to exactly that. + +Runs in two different / wrappers are never merged: that would +rewrite tracked-change structure, collapsing separate revisions into one. + +Only word/document.xml is processed (not headers, footers, or footnotes). + +Usage: + python merge_runs.py unpacked/ # after unzip, before editing + python merge_runs.py document.docx # rewrite in place + python merge_runs.py document.docx -o out.docx +""" + + +import argparse +import sys +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.minidom + +from office.helpers import XML_SPACE, rendered_text, rezip, safe_extract + +WORDML_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + + +def merge_runs(input_dir: str) -> tuple[int, str]: + doc_xml = Path(input_dir) / "word" / "document.xml" + + if not doc_xml.exists(): + return 0, f"Error: {doc_xml} not found" + + try: + dom = defusedxml.minidom.parseString(doc_xml.read_text(encoding="utf-8")) + root = dom.documentElement + run_names = _run_tag_names(root) + + _remove_elements(root, "proofErr") + + runs = _find_runs(root, run_names) + _strip_rsid_attrs(runs) + + merge_count = 0 + for container in {run.parentNode for run in runs}: + merge_count += _merge_runs_in(container, run_names) + + doc_xml.write_bytes(dom.toxml(encoding="UTF-8")) + return merge_count, f"Merged {merge_count} runs" + + except Exception as e: + return 0, f"Error: {e}" + + + + +def _is_element(node, tag: str) -> bool: + name = node.localName or node.tagName + return name == tag or name.endswith(f":{tag}") + + +def _run_tag_names(root) -> set[str]: + names = set() + for attr in root.attributes.values(): + if attr.value == WORDML_NS: + if attr.name == "xmlns": + names.add("r") + elif attr.name.startswith("xmlns:"): + names.add(attr.name.split(":", 1)[1] + ":r") + return names or {"w:r", "r"} + + +def _find_elements(root, tag: str) -> list: + results = [] + + def traverse(node): + if node.nodeType == node.ELEMENT_NODE: + if _is_element(node, tag): + results.append(node) + for child in node.childNodes: + traverse(child) + + traverse(root) + return results + + +def _find_runs(root, run_names: set[str]) -> list: + return [e for e in _find_elements(root, "r") if _is_run(e, run_names)] + + +def _get_child(parent, tag: str): + return next(iter(_get_children(parent, tag)), None) + + +def _get_children(parent, tag: str) -> list: + return [ + child + for child in parent.childNodes + if child.nodeType == child.ELEMENT_NODE and _is_element(child, tag) + ] + + +def _is_adjacent(elem1, elem2) -> bool: + node = elem1.nextSibling + while node: + if node == elem2: + return True + if node.nodeType == node.ELEMENT_NODE: + return False + if node.nodeType == node.TEXT_NODE and node.data.strip(XML_SPACE): + return False + node = node.nextSibling + return False + + + + +def _remove_elements(root, tag: str): + for elem in _find_elements(root, tag): + if elem.parentNode: + elem.parentNode.removeChild(elem) + + +def _strip_rsid_attrs(runs: list): + for run in runs: + for attr in list(run.attributes.values()): + if "rsid" in attr.name.lower(): + run.removeAttribute(attr.name) + + + + +def _merge_runs_in(container, run_names: set[str]) -> int: + merge_count = 0 + run = _first_child_run(container, run_names) + + while run: + while True: + next_elem = _next_element_sibling(run) + if next_elem and _is_run(next_elem, run_names) and _can_merge(run, next_elem): + _merge_run_content(run, next_elem) + container.removeChild(next_elem) + merge_count += 1 + else: + break + + _consolidate_text(run) + run = _next_sibling_run(run, run_names) + + return merge_count + + +def _first_child_run(container, run_names: set[str]): + for child in container.childNodes: + if child.nodeType == child.ELEMENT_NODE and _is_run(child, run_names): + return child + return None + + +def _next_element_sibling(node): + sibling = node.nextSibling + while sibling: + if sibling.nodeType == sibling.ELEMENT_NODE: + return sibling + sibling = sibling.nextSibling + return None + + +def _next_sibling_run(node, run_names: set[str]): + sibling = node.nextSibling + while sibling: + if sibling.nodeType == sibling.ELEMENT_NODE: + if _is_run(sibling, run_names): + return sibling + sibling = sibling.nextSibling + return None + + +def _is_run(node, run_names: set[str]) -> bool: + return node.tagName in run_names + + +def _can_merge(run1, run2) -> bool: + rpr1 = _get_child(run1, "rPr") + rpr2 = _get_child(run2, "rPr") + + if (rpr1 is None) != (rpr2 is None): + return False + if rpr1 is None: + return True + return rpr1.toxml() == rpr2.toxml() + + +def _merge_run_content(target, source): + for child in list(source.childNodes): + if child.nodeType == child.ELEMENT_NODE: + name = child.localName or child.tagName + if name != "rPr" and not name.endswith(":rPr"): + target.appendChild(child) + + +def _element_text(elem) -> str: + return "".join( + child.data + for child in elem.childNodes + if child.nodeType in (child.TEXT_NODE, child.CDATA_SECTION_NODE) + ) + + +def _has_preserve(elem) -> bool: + return elem.getAttribute("xml:space") == "preserve" + + +def _rendered_text(elem) -> str: + return rendered_text(_element_text(elem), _has_preserve(elem)) + + +def _consolidate_text(run): + for tag in ("t", "delText"): + _consolidate_text_elements(run, tag) + + +def _consolidate_text_elements(run, tag: str): + t_elements = _get_children(run, tag) + + for i in range(len(t_elements) - 1, 0, -1): + curr, prev = t_elements[i], t_elements[i - 1] + + if _is_adjacent(prev, curr): + merged = _rendered_text(prev) + _rendered_text(curr) + had_preserve = _has_preserve(prev) or _has_preserve(curr) + + new_text = run.ownerDocument.createTextNode(merged) + for node in list(prev.childNodes): + if node.nodeType in (node.TEXT_NODE, node.CDATA_SECTION_NODE): + prev.removeChild(node) + else: + run.insertBefore(node, curr) + prev.appendChild(new_text) + for node in list(curr.childNodes): + if node.nodeType not in (node.TEXT_NODE, node.CDATA_SECTION_NODE): + run.insertBefore(node, curr) + + if merged != merged.strip(XML_SPACE) or had_preserve: + prev.setAttribute("xml:space", "preserve") + elif prev.hasAttribute("xml:space"): + prev.removeAttribute("xml:space") + + run.removeChild(curr) + + + + +def _merge_or_die(path: Path) -> str: + _, msg = merge_runs(str(path)) + if msg.startswith("Error"): + print(msg, file=sys.stderr) + sys.exit(1) + return msg + + +def main() -> None: + p = argparse.ArgumentParser( + description="Merge adjacent identically-formatted runs in a DOCX (directory or .docx file)." + ) + p.add_argument("input", help="Unpacked DOCX directory OR a .docx/.dotx file") + p.add_argument( + "-o", "--output", + help="Output .docx path (only valid when input is a .docx; default: overwrite input)", + ) + args = p.parse_args() + + src = Path(args.input) + + try: + if src.is_dir(): + if args.output: + p.error("--output is only valid for .docx input; directory input is modified in place") + print(_merge_or_die(src)) + elif src.is_file() and src.suffix.lower() in (".docx", ".dotx"): + out = Path(args.output) if args.output else src + with tempfile.TemporaryDirectory() as tmp: + tmp_path = Path(tmp) + with zipfile.ZipFile(src) as zf: + safe_extract(zf, tmp_path) + msg = _merge_or_die(tmp_path) + rezip(tmp_path, out) + print(f"{msg}; wrote {out}") + else: + print(f"Error: {src} is neither a directory nor a .docx/.dotx file", file=sys.stderr) + sys.exit(1) + except (OSError, ValueError, zipfile.BadZipFile) as e: + print(f"Error: {e}", file=sys.stderr) + sys.exit(1) + + +if __name__ == "__main__": + main() diff --git a/.github/skills/anthropic-docx/scripts/office/helpers/__init__.py b/.github/skills/anthropic-docx/scripts/office/helpers/__init__.py new file mode 100644 index 00000000..d3c5817c --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/helpers/__init__.py @@ -0,0 +1,111 @@ +import os +import posixpath +import re +import stat +import tempfile +import urllib.parse +import zipfile +from pathlib import Path + +OOXML_FAMILY = { + ".docx": "docx", + ".dotx": "docx", + ".pptx": "pptx", + ".potx": "pptx", + ".xlsx": "xlsx", + ".xltx": "xlsx", +} + +_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.\-]*:") + +SLIDE_REL_TYPE = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/slide" + + +def opc_target(target: str, source_part: str, target_mode: str = "") -> str | None: + if not target: + return None + if target_mode.lower() == "external": + return None + if _SCHEME_RE.match(target): + return None + + target = urllib.parse.unquote(target) + + if "\\" in target: + raise ValueError(f"relationship target is not a POSIX part name: {target!r}") + + if target.startswith("/"): + joined = target.lstrip("/") + else: + joined = posixpath.join(posixpath.dirname(source_part), target) + + parts: list[str] = [] + for segment in posixpath.normpath(joined).split("/"): + if segment in ("", "."): + continue + if segment == "..": + if not parts: + raise ValueError(f"relationship target escapes the package: {target!r}") + parts.pop() + else: + parts.append(segment) + + if not parts: + raise ValueError(f"relationship target resolves to nothing: {target!r}") + return "/".join(parts) + + +def rels_source_part(rels_file: Path, unpacked_dir: Path) -> str: + owner_dir = rels_file.parent.parent.relative_to(unpacked_dir) + return posixpath.join(owner_dir.as_posix(), rels_file.name[: -len(".rels")]).lstrip("./") + + +def part_text(data: bytes) -> str: + return data.decode("utf-8", "surrogateescape") + + +XML_SPACE = " \t\r\n" + + +def rendered_text(text: str, preserve: bool) -> str: + return text if preserve else text.strip(XML_SPACE) + + +def safe_extract(zf: zipfile.ZipFile, dest: Path) -> None: + dest = dest.resolve() + for m in zf.infolist(): + if stat.S_ISLNK(m.external_attr >> 16): + raise ValueError(f"symlink archive entry not allowed: {m.filename!r}") + target = (dest / m.filename).resolve() + if not target.is_relative_to(dest): + raise ValueError(f"unsafe archive entry: {m.filename!r}") + zf.extract(m, dest) + + +def rezip(src_dir: Path, out_path: Path) -> None: + files = sorted(p for p in src_dir.rglob("*") if p.is_file()) + ct = src_dir / "[Content_Types].xml" + fd, tmp_name = tempfile.mkstemp( + prefix=out_path.name + ".", suffix=".tmp", dir=out_path.parent + ) + tmp_out = Path(tmp_name) + try: + with os.fdopen(fd, "wb") as fh: + with zipfile.ZipFile(fh, "w", zipfile.ZIP_DEFLATED) as zf: + if ct.exists(): + zf.write(ct, ct.relative_to(src_dir), compress_type=zipfile.ZIP_STORED) + for f in files: + if f == ct: + continue + zf.write(f, f.relative_to(src_dir)) + if out_path.exists(): + mode = out_path.stat().st_mode & 0o777 + else: + umask = os.umask(0) + os.umask(umask) + mode = 0o666 & ~umask + os.chmod(tmp_out, mode) + os.replace(tmp_out, out_path) + finally: + if tmp_out.exists(): + tmp_out.unlink() diff --git a/.github/skills/anthropic-docx/scripts/office/helpers/pptx_chart.py b/.github/skills/anthropic-docx/scripts/office/helpers/pptx_chart.py new file mode 100644 index 00000000..209cb7c5 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/helpers/pptx_chart.py @@ -0,0 +1,170 @@ +"""Find chart XML that PowerPoint refuses but the schema accepts. + +Detection only: for either fault more than one repair is valid, and only the +author knows which was meant. +""" + + +from __future__ import annotations + +import re +from typing import Mapping + +from . import part_text + + +_CHART_PART_RE = re.compile(r"ppt/charts/chart\d+\.xml") + +_GROUPING_RE = re.compile(r"""]*?\bval=["'](\w+)["']""") +_DLBL_POS_RE = re.compile(r"""]*?\bval=["'](\w+)["']""") + +def _strip_ext_lst(text: str) -> str: + out, cursor = [], 0 + for lo, hi in _ext_lst_spans(text): + out.append(text[cursor:lo]) + cursor = hi + out.append(text[cursor:]) + return "".join(out) + +_BAR_GROUP_RE = re.compile(r"]*(?.*?", re.DOTALL) + +STACKED_GROUPINGS = frozenset({"stacked", "percentStacked"}) +ILLEGAL_ON_STACKED = frozenset({"outEnd"}) +LEGAL_ON_STACKED = ("ctr", "inEnd", "inBase") + + +def _check_stacked_label_positions(part: str, xml: str) -> list[str]: + problems: list[str] = [] + for match in _BAR_GROUP_RE.finditer(xml): + block = _strip_ext_lst(match.group(0)) + group = match.group(1) + + grouping = _GROUPING_RE.search(block) + if grouping is None or grouping.group(1) not in STACKED_GROUPINGS: + continue + + bad = [p for p in _DLBL_POS_RE.findall(block) if p in ILLEGAL_ON_STACKED] + for pos in sorted(set(bad)): + problems.append( + f'{part}: {bad.count(pos)} data label(s) use dLblPos="{pos}" on a ' + f"{grouping.group(1)} {group}; PowerPoint allows only " + f"{', '.join(LEGAL_ON_STACKED)} there" + ) + return problems + + + +_ANY_CHART_GROUP_RE = re.compile(r"]*(?.*?", re.DOTALL) + +_AXID_RE = re.compile( + r"""\s*]*?\bval=["'](-?\d+)["']\s*(?:/>|>\s*)""" +) + +_AXIS_DECL_RE = re.compile( + r"""]*(?\s*]*?\bval=["'](-?\d+)["']""" +) + +AXID_LIMIT = { + "barChart": 2, "lineChart": 2, "areaChart": 2, "scatterChart": 2, + "bubbleChart": 2, "radarChart": 2, "stockChart": 2, + "bar3DChart": 3, "line3DChart": 3, "area3DChart": 3, + "surfaceChart": 3, "surface3DChart": 3, +} + +AXID_MINIMUM = { + "barChart": 2, "lineChart": 2, "areaChart": 2, "scatterChart": 2, + "bubbleChart": 2, "radarChart": 2, "stockChart": 2, + "bar3DChart": 2, "area3DChart": 2, "surfaceChart": 2, + "line3DChart": 3, "surface3DChart": 3, +} + + +def _declared_axes(xml: str) -> dict[str, list[str]]: + axes: dict[str, list[str]] = {} + for kind, axid in _AXIS_DECL_RE.findall(xml): + axes.setdefault(kind, []).append(axid) + return axes + + +def _canonical_ids(axes: dict[str, list[str]], limit: int) -> list[str] | None: + category = axes.get("catAx", []) + axes.get("dateAx", []) + value = axes.get("valAx", []) + series = axes.get("serAx", []) + if len(category) != 1 or len(value) != 1 or len(series) > 1: + return None + ids = [category[0], value[0]] + if limit >= 3 and series: + ids.append(series[0]) + return ids + + +def _undeclared_axes(kind: str, block: str, axes: dict[str, list[str]]) -> list[str] | None: + if kind not in AXID_LIMIT: + return None + ids = _AXID_RE.findall(block) + declared = {i for group in axes.values() for i in group} + if len([i for i in ids if i in declared]) >= 2: + return None + return ids + + +def _check_chart_axis_references(part: str, xml: str) -> list[str]: + axes = _declared_axes(xml) + problems: list[str] = [] + declared = {i for group in axes.values() for i in group} + for match in _ANY_CHART_GROUP_RE.finditer(xml): + kind, block = match.group(1), match.group(0) + ids = _undeclared_axes(kind, block, axes) + if ids is None: + continue + if not ids: + problems.append( + f"{part}: declares no this part can resolve; a chart " + f"group needs {AXID_MINIMUM[kind]}, and PowerPoint discards one with fewer" + ) + continue + dead = [i for i in ids if i not in declared] + canonical = _canonical_ids(axes, AXID_LIMIT[kind]) + if canonical is not None and len(canonical) >= AXID_MINIMUM[kind]: + hint = f"Fix: point them at the axes this part declares ({', '.join(canonical)})" + else: + hint = ("Fix: the part declares several axes of a kind -- declare the " + "secondary axes the series expects, or drop them") + detail = (f"of which {', '.join(dead)} name no declared axis" + if dead else f"only {len(ids)} of which this part declares") + problems.append( + f"{part}: references axId {', '.join(ids)}, {detail}, " + f"leaving fewer than two live axes; PowerPoint discards the chart. {hint}" + ) + return problems + + +def _ext_lst_spans(text: str) -> list[tuple[int, int]]: + spans: list[tuple[int, int]] = [] + depth = 0 + start = 0 + for match in re.finditer(r"<(/?)c:extLst\b[^>]*?(/?)>", text): + closing, self_closing = match.group(1), match.group(2) + if self_closing: + continue + if closing: + depth -= 1 + if depth == 0: + spans.append((start, match.end())) + else: + if depth == 0: + start = match.start() + depth += 1 + return spans + + +CHART_CHECKS = (_check_stacked_label_positions, _check_chart_axis_references) + + +def find_chart_problems(files: Mapping[str, bytes]) -> list[str]: + problems: list[str] = [] + for part in sorted(n for n in files if _CHART_PART_RE.fullmatch(n)): + xml = part_text(files[part]) + for check in CHART_CHECKS: + problems.extend(check(part, xml)) + return problems diff --git a/.github/skills/anthropic-docx/scripts/office/helpers/pptx_slide.py b/.github/skills/anthropic-docx/scripts/office/helpers/pptx_slide.py new file mode 100644 index 00000000..22f9aee0 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/helpers/pptx_slide.py @@ -0,0 +1,60 @@ +"""Pick the slide-XML schema errors PowerPoint refuses the file over. + +A denylist over lxml's messages, so an unrecognised error class is a miss rather +than a false alarm. +""" + + +from __future__ import annotations + +import re + +SLIDE_PART_RE = re.compile( + r"ppt/(slides|slideLayouts|slideMasters|notesSlides|notesMasters|handoutMasters)" + r"/[^/]+\.xml" +) + +FATAL_SLIDE_ERRORS: tuple[tuple[re.Pattern[str], str], ...] = ( + ( + re.compile(r"\}tableStyleId': This element is not expected"), + "two in one (the schema allows one)", + ), + ( + re.compile(r"\}srgbClr', attribute 'val'"), + "a colour that is not six hex digits", + ), + ( + re.compile(r"\}txBody': Missing child element"), + "a with no children", + ), + ( + re.compile(r"\}miter', attribute 'lim'"), + 'a line join with lim="NaN"', + ), + ( + re.compile(r"\}uLnTx': This element is not expected"), + " in a position the schema forbids", + ), + ( + re.compile(r"\}overrideClrMapping': This element is not expected"), + " in a position the schema forbids", + ), + ( + re.compile(r"\}nvGrpSpPr': Missing child element"), + "a with no children", + ), +) + + +def is_schema_verdict(error: str) -> bool: + return error.startswith("Element ") + + +def fatal_slide_errors(errors: set[str]) -> list[str]: + out = [] + for error in sorted(errors): + for pattern, meaning in FATAL_SLIDE_ERRORS: + if pattern.search(error): + out.append(f"{meaning}: {error}") + break + return out diff --git a/.github/skills/anthropic-docx/scripts/office/helpers/pptx_theme.py b/.github/skills/anthropic-docx/scripts/office/helpers/pptx_theme.py new file mode 100644 index 00000000..84466201 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/helpers/pptx_theme.py @@ -0,0 +1,114 @@ +"""Find masters sharing a theme part in the way PowerPoint refuses to open. + +Reports only; the fix is to move back to directly after + in ppt/presentation.xml. +""" + + +from __future__ import annotations + +import posixpath +import re +from typing import Mapping + +from . import part_text + +THEME_REL_TYPE = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/theme" + +_MASTER_RE = re.compile( + r"^ppt/(?PslideMasters|notesMasters|handoutMasters)/" + r"(?:slide|notes|handout)Master(?P\d+)\.xml$" +) +_GROUP_ORDER = {"slideMasters": 0, "notesMasters": 1, "handoutMasters": 2} + +_RELATIONSHIP_RE = re.compile( + r"]*?(?:/>|>.*?)", re.DOTALL +) + + +def _sort_key(name: str) -> tuple[int, int]: + m = _MASTER_RE.match(name) + assert m is not None + return (_GROUP_ORDER[m.group("group")], int(m.group("num"))) + + +def _rels_path(part: str) -> str: + directory, base = posixpath.split(part) + return f"{directory}/_rels/{base}.rels" + + +def _resolve(rels_path: str, target: str) -> str: + if target.startswith("/"): + return target.lstrip("/") + part_dir = posixpath.dirname(posixpath.dirname(rels_path)) + return posixpath.normpath(posixpath.join(part_dir, target)) + + +def _theme_rel(files: Mapping[str, bytes], master: str): + rels_path = _rels_path(master) + rels = files.get(rels_path) + if rels is None: + return None + for element in _RELATIONSHIP_RE.findall(part_text(rels)): + if f'Type="{THEME_REL_TYPE}"' not in element: + continue + target = re.search(r'\bTarget="([^"]+)"', element) + if target is None: + continue + return rels_path, element, _resolve(rels_path, target.group(1)) + return None + + +def _masters(files: Mapping[str, bytes]) -> list[str]: + return sorted((n for n in files if _MASTER_RE.match(n)), key=_sort_key) + + +_PRESENTATION = "ppt/presentation.xml" +_NOTES_MASTERS = "ppt/notesMasters/" +_IGNORABLE_RE = re.compile(r"|<\?.*?\?>", re.DOTALL) +_AFTER_SLDIDLST_RE = re.compile( + r"]*/>|[^>]*>.*?)\s*(<[^>\s/]+)", re.DOTALL +) + + +def _notes_master_share_is_inert(files: Mapping[str, bytes]) -> bool: + data = files.get(_PRESENTATION) + if data is None: + return False + match = _AFTER_SLDIDLST_RE.search(_IGNORABLE_RE.sub("", part_text(data))) + return match is not None and match.group(1) == " bool: + return inert_notes and master.startswith(_NOTES_MASTERS) + + +def find_shared_master_themes(files: Mapping[str, bytes]) -> list[str]: + return [ + f"{master} shares {theme} with {first}" + for master, _, _, theme, first in _shares(files) + ] + + +def live_shared_master_themes(files: Mapping[str, bytes]) -> list[str]: + inert_notes = _notes_master_share_is_inert(files) + return [ + f"{master} shares {theme} with {first}" + for master, _, _, theme, first in _shares(files) + if not _is_inert(master, inert_notes) + ] diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chart.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chart.xsd new file mode 100644 index 00000000..6454ef9a --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chart.xsd @@ -0,0 +1,1499 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chartDrawing.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chartDrawing.xsd new file mode 100644 index 00000000..afa4f463 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chartDrawing.xsd @@ -0,0 +1,146 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-diagram.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-diagram.xsd new file mode 100644 index 00000000..64e66b8a --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-diagram.xsd @@ -0,0 +1,1085 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-lockedCanvas.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-lockedCanvas.xsd new file mode 100644 index 00000000..687eea82 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-lockedCanvas.xsd @@ -0,0 +1,11 @@ + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-main.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-main.xsd new file mode 100644 index 00000000..6ac81b06 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-main.xsd @@ -0,0 +1,3081 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-picture.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-picture.xsd new file mode 100644 index 00000000..1dbf0514 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-picture.xsd @@ -0,0 +1,23 @@ + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-spreadsheetDrawing.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-spreadsheetDrawing.xsd new file mode 100644 index 00000000..f1af17db --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-spreadsheetDrawing.xsd @@ -0,0 +1,185 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-wordprocessingDrawing.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-wordprocessingDrawing.xsd new file mode 100644 index 00000000..0a185ab6 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-wordprocessingDrawing.xsd @@ -0,0 +1,287 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/pml.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/pml.xsd new file mode 100644 index 00000000..14ef4888 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/pml.xsd @@ -0,0 +1,1676 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-additionalCharacteristics.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-additionalCharacteristics.xsd new file mode 100644 index 00000000..c20f3bf1 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-additionalCharacteristics.xsd @@ -0,0 +1,28 @@ + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-bibliography.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-bibliography.xsd new file mode 100644 index 00000000..ac602522 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-bibliography.xsd @@ -0,0 +1,144 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-commonSimpleTypes.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-commonSimpleTypes.xsd new file mode 100644 index 00000000..424b8ba8 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-commonSimpleTypes.xsd @@ -0,0 +1,174 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlDataProperties.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlDataProperties.xsd new file mode 100644 index 00000000..2bddce29 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlDataProperties.xsd @@ -0,0 +1,25 @@ + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlSchemaProperties.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlSchemaProperties.xsd new file mode 100644 index 00000000..8a8c18ba --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlSchemaProperties.xsd @@ -0,0 +1,18 @@ + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd new file mode 100644 index 00000000..5c42706a --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd @@ -0,0 +1,59 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd new file mode 100644 index 00000000..853c341c --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd @@ -0,0 +1,56 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesVariantTypes.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesVariantTypes.xsd new file mode 100644 index 00000000..da835ee8 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesVariantTypes.xsd @@ -0,0 +1,195 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-math.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-math.xsd new file mode 100644 index 00000000..87ad2658 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-math.xsd @@ -0,0 +1,582 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-relationshipReference.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-relationshipReference.xsd new file mode 100644 index 00000000..9e86f1b2 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-relationshipReference.xsd @@ -0,0 +1,25 @@ + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/sml.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/sml.xsd new file mode 100644 index 00000000..d0be42e7 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/sml.xsd @@ -0,0 +1,4439 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-main.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-main.xsd new file mode 100644 index 00000000..8821dd18 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-main.xsd @@ -0,0 +1,570 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-officeDrawing.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-officeDrawing.xsd new file mode 100644 index 00000000..ca2575c7 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-officeDrawing.xsd @@ -0,0 +1,509 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-presentationDrawing.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-presentationDrawing.xsd new file mode 100644 index 00000000..dd079e60 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-presentationDrawing.xsd @@ -0,0 +1,12 @@ + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-spreadsheetDrawing.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-spreadsheetDrawing.xsd new file mode 100644 index 00000000..3dd6cf62 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-spreadsheetDrawing.xsd @@ -0,0 +1,108 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-wordprocessingDrawing.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-wordprocessingDrawing.xsd new file mode 100644 index 00000000..f1041e34 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-wordprocessingDrawing.xsd @@ -0,0 +1,96 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/wml.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/wml.xsd new file mode 100644 index 00000000..9c5b7a63 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/wml.xsd @@ -0,0 +1,3646 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/xml.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/xml.xsd new file mode 100644 index 00000000..0f13678d --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ISO-IEC29500-4_2016/xml.xsd @@ -0,0 +1,116 @@ + + + + + + See http://www.w3.org/XML/1998/namespace.html and + http://www.w3.org/TR/REC-xml for information about this namespace. + + This schema document describes the XML namespace, in a form + suitable for import by other schema documents. + + Note that local names in this namespace are intended to be defined + only by the World Wide Web Consortium or its subgroups. The + following names are currently defined in this namespace and should + not be used with conflicting semantics by any Working Group, + specification, or document instance: + + base (as an attribute name): denotes an attribute whose value + provides a URI to be used as the base for interpreting any + relative URIs in the scope of the element on which it + appears; its value is inherited. This name is reserved + by virtue of its definition in the XML Base specification. + + lang (as an attribute name): denotes an attribute whose value + is a language code for the natural language of the content of + any element; its value is inherited. This name is reserved + by virtue of its definition in the XML specification. + + space (as an attribute name): denotes an attribute whose + value is a keyword indicating what whitespace processing + discipline is intended for the content of the element; its + value is inherited. This name is reserved by virtue of its + definition in the XML specification. + + Father (in any context at all): denotes Jon Bosak, the chair of + the original XML Working Group. This name is reserved by + the following decision of the W3C XML Plenary and + XML Coordination groups: + + In appreciation for his vision, leadership and dedication + the W3C XML Plenary on this 10th day of February, 2000 + reserves for Jon Bosak in perpetuity the XML name + xml:Father + + + + + This schema defines attributes and an attribute group + suitable for use by + schemas wishing to allow xml:base, xml:lang or xml:space attributes + on elements they define. + + To enable this, such a schema must import this schema + for the XML namespace, e.g. as follows: + <schema . . .> + . . . + <import namespace="http://www.w3.org/XML/1998/namespace" + schemaLocation="http://www.w3.org/2001/03/xml.xsd"/> + + Subsequently, qualified reference to any of the attributes + or the group defined below will have the desired effect, e.g. + + <type . . .> + . . . + <attributeGroup ref="xml:specialAttrs"/> + + will define a type which will schema-validate an instance + element with any of those attributes + + + + In keeping with the XML Schema WG's standard versioning + policy, this schema document will persist at + http://www.w3.org/2001/03/xml.xsd. + At the date of issue it can also be found at + http://www.w3.org/2001/xml.xsd. + The schema document at that URI may however change in the future, + in order to remain compatible with the latest version of XML Schema + itself. In other words, if the XML Schema namespace changes, the version + of this document at + http://www.w3.org/2001/xml.xsd will change + accordingly; the version at + http://www.w3.org/2001/03/xml.xsd will not change. + + + + + + In due course, we should install the relevant ISO 2- and 3-letter + codes as the enumerated possible values . . . + + + + + + + + + + + + + + + See http://www.w3.org/TR/xmlbase/ for + information about this attribute. + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-contentTypes.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-contentTypes.xsd new file mode 100644 index 00000000..a6de9d27 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-contentTypes.xsd @@ -0,0 +1,42 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-coreProperties.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-coreProperties.xsd new file mode 100644 index 00000000..10e978b6 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-coreProperties.xsd @@ -0,0 +1,50 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-digSig.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-digSig.xsd new file mode 100644 index 00000000..4248bf7a --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-digSig.xsd @@ -0,0 +1,49 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-relationships.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-relationships.xsd new file mode 100644 index 00000000..56497467 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/ecma/fouth-edition/opc-relationships.xsd @@ -0,0 +1,33 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/mce/mc.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/mce/mc.xsd new file mode 100644 index 00000000..ef725457 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/mce/mc.xsd @@ -0,0 +1,75 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-2010.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-2010.xsd new file mode 100644 index 00000000..f65f7777 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-2010.xsd @@ -0,0 +1,560 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-2012.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-2012.xsd new file mode 100644 index 00000000..6b00755a --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-2012.xsd @@ -0,0 +1,67 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-2018.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-2018.xsd new file mode 100644 index 00000000..f321d333 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-2018.xsd @@ -0,0 +1,14 @@ + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-cex-2018.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-cex-2018.xsd new file mode 100644 index 00000000..364c6a9b --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-cex-2018.xsd @@ -0,0 +1,20 @@ + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-cid-2016.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-cid-2016.xsd new file mode 100644 index 00000000..fed9d15b --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-cid-2016.xsd @@ -0,0 +1,13 @@ + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-sdtdatahash-2020.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-sdtdatahash-2020.xsd new file mode 100644 index 00000000..680cf154 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-sdtdatahash-2020.xsd @@ -0,0 +1,4 @@ + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-symex-2015.xsd b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-symex-2015.xsd new file mode 100644 index 00000000..89ada908 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/schemas/microsoft/wml-symex-2015.xsd @@ -0,0 +1,8 @@ + + + + + + + + diff --git a/.github/skills/anthropic-docx/scripts/office/soffice.py b/.github/skills/anthropic-docx/scripts/office/soffice.py new file mode 100644 index 00000000..0b4c99de --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/soffice.py @@ -0,0 +1,192 @@ +""" +Helper for running LibreOffice (soffice) in environments where AF_UNIX +sockets may be blocked (e.g., sandboxed VMs). Detects the restriction +at runtime and applies an LD_PRELOAD shim if needed. + +Usage: + from office.soffice import run_soffice + + result = run_soffice(["--headless", "--convert-to", "pdf", "input.docx"]) + +Call soffice through run_soffice, not through subprocess with get_soffice_env(): +the env dict carries the shim but names no user profile, and a non-root sandbox +cannot bootstrap the default one -- soffice aborts with "User installation could +not be completed" and converts nothing. get_soffice_env() stays public for the +callers that build their own argv (they must pass -env:UserInstallation too). +""" + +import contextlib +import os +import socket +import subprocess +import tempfile +from collections.abc import Iterable +from pathlib import Path + + +def get_soffice_env() -> dict: + env = os.environ.copy() + env["SAL_USE_VCLPLUGIN"] = "svp" + + if _needs_shim(): + shim = _ensure_shim() + env["LD_PRELOAD"] = str(shim) + + return env + + +def run_soffice(args: Iterable[str], **kwargs) -> subprocess.CompletedProcess: + args = list(args) + with contextlib.ExitStack() as stack: + if not any(str(a).startswith("-env:UserInstallation") for a in args): + profile = stack.enter_context( + tempfile.TemporaryDirectory(prefix="lo_profile_", ignore_cleanup_errors=True) + ) + args = [f"-env:UserInstallation={Path(profile).as_uri()}"] + args + return subprocess.run(["soffice"] + args, env=get_soffice_env(), **kwargs) + + + +_SHIM_SO = Path(tempfile.gettempdir()) / "lo_socket_shim.so" + + +def _needs_shim() -> bool: + try: + s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + s.close() + return False + except OSError: + return True + + +def _ensure_shim() -> Path: + if _SHIM_SO.exists(): + return _SHIM_SO + + src = Path(tempfile.gettempdir()) / "lo_socket_shim.c" + src.write_text(_SHIM_SOURCE) + subprocess.run( + ["gcc", "-shared", "-fPIC", "-o", str(_SHIM_SO), str(src), "-ldl"], + check=True, + capture_output=True, + ) + src.unlink() + return _SHIM_SO + + + +_SHIM_SOURCE = r""" +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include + +static int (*real_socket)(int, int, int); +static int (*real_socketpair)(int, int, int, int[2]); +static int (*real_listen)(int, int); +static int (*real_accept)(int, struct sockaddr *, socklen_t *); +static int (*real_close)(int); +static int (*real_read)(int, void *, size_t); + +/* Per-FD bookkeeping (FDs >= 1024 are passed through unshimmed). */ +static int is_shimmed[1024]; +static int peer_of[1024]; +static int wake_r[1024]; /* accept() blocks reading this */ +static int wake_w[1024]; /* close() writes to this */ +static int listener_fd = -1; /* FD that received listen() */ + +__attribute__((constructor)) +static void init(void) { + real_socket = dlsym(RTLD_NEXT, "socket"); + real_socketpair = dlsym(RTLD_NEXT, "socketpair"); + real_listen = dlsym(RTLD_NEXT, "listen"); + real_accept = dlsym(RTLD_NEXT, "accept"); + real_close = dlsym(RTLD_NEXT, "close"); + real_read = dlsym(RTLD_NEXT, "read"); + for (int i = 0; i < 1024; i++) { + peer_of[i] = -1; + wake_r[i] = -1; + wake_w[i] = -1; + } +} + +/* ---- socket ---------------------------------------------------------- */ +int socket(int domain, int type, int protocol) { + if (domain == AF_UNIX) { + int fd = real_socket(domain, type, protocol); + if (fd >= 0) return fd; + /* socket(AF_UNIX) blocked – fall back to socketpair(). */ + int sv[2]; + if (real_socketpair(domain, type, protocol, sv) == 0) { + if (sv[0] >= 0 && sv[0] < 1024) { + is_shimmed[sv[0]] = 1; + peer_of[sv[0]] = sv[1]; + int wp[2]; + if (pipe(wp) == 0) { + wake_r[sv[0]] = wp[0]; + wake_w[sv[0]] = wp[1]; + } + } + return sv[0]; + } + errno = EPERM; + return -1; + } + return real_socket(domain, type, protocol); +} + +/* ---- listen ---------------------------------------------------------- */ +int listen(int sockfd, int backlog) { + if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) { + listener_fd = sockfd; + return 0; + } + return real_listen(sockfd, backlog); +} + +/* ---- accept ---------------------------------------------------------- */ +int accept(int sockfd, struct sockaddr *addr, socklen_t *addrlen) { + if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) { + /* Block until close() writes to the wake pipe. */ + if (wake_r[sockfd] >= 0) { + char buf; + real_read(wake_r[sockfd], &buf, 1); + } + errno = ECONNABORTED; + return -1; + } + return real_accept(sockfd, addr, addrlen); +} + +/* ---- close ----------------------------------------------------------- */ +int close(int fd) { + if (fd >= 0 && fd < 1024 && is_shimmed[fd]) { + int was_listener = (fd == listener_fd); + is_shimmed[fd] = 0; + + if (wake_w[fd] >= 0) { /* unblock accept() */ + char c = 0; + write(wake_w[fd], &c, 1); + real_close(wake_w[fd]); + wake_w[fd] = -1; + } + if (wake_r[fd] >= 0) { real_close(wake_r[fd]); wake_r[fd] = -1; } + if (peer_of[fd] >= 0) { real_close(peer_of[fd]); peer_of[fd] = -1; } + + if (was_listener) + _exit(0); /* conversion done – exit */ + } + return real_close(fd); +} +""" + + + +if __name__ == "__main__": + import sys + result = run_soffice(sys.argv[1:]) + sys.exit(result.returncode) diff --git a/.github/skills/anthropic-docx/scripts/office/validate.py b/.github/skills/anthropic-docx/scripts/office/validate.py new file mode 100644 index 00000000..29ca186a --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/validate.py @@ -0,0 +1,173 @@ +""" +Command line tool to validate Office document XML files against XSD schemas and tracked changes. + +Usage: + python validate.py [--original ] [--auto-repair] [--author NAME] + +The first argument can be either: +- An unpacked directory containing the Office document XML files +- A packed Office file (.docx/.pptx/.xlsx or .dotx/.potx/.xltx template) which will be unpacked to a temp directory + +Auto-repair fixes: +- paraId/durableId values that exceed OOXML limits +- Missing xml:space="preserve" on w:t elements with whitespace +""" + +import argparse +import sys +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.ElementTree as ET +from defusedxml.common import DefusedXmlException + +from helpers import OOXML_FAMILY, rezip, safe_extract +from validators import DOCXSchemaValidator, PPTXSchemaValidator, RedliningValidator + +WORD_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + + +def _fail(message: str): + print(f"Error: {message}", file=sys.stderr) + sys.exit(2) + + +def _has_tracked_changes(unpacked_dir: Path) -> bool: + document = unpacked_dir / "word" / "document.xml" + if not document.is_file(): + return False + try: + root = ET.parse(document).getroot() + except (ET.ParseError, DefusedXmlException): + return False + tracked = {f"{{{WORD_NS}}}ins", f"{{{WORD_NS}}}del"} + return any(elem.tag in tracked for elem in root.iter()) + + +def main(): + parser = argparse.ArgumentParser(description="Validate Office document XML files") + parser.add_argument( + "path", + help="Path to unpacked directory or packed Office file (.docx/.pptx/.xlsx or .dotx/.potx/.xltx)", + ) + parser.add_argument( + "--original", + required=False, + default=None, + help="Path to original file (.docx/.pptx/.xlsx or .dotx/.potx/.xltx). If omitted, all XSD errors are reported and redlining validation is skipped.", + ) + parser.add_argument( + "-v", + "--verbose", + action="store_true", + help="Enable verbose output", + ) + parser.add_argument( + "--auto-repair", + action="store_true", + help="Automatically repair common issues (hex IDs, whitespace preservation). " + "Modifies the input in place: repairs to a packed file are written back to it.", + ) + parser.add_argument( + "--author", + default=None, + help="The name you are redlining under. Passing it turns on the " + "tracked-change check: any text differing from --original without a " + "/ recording it is reported. Untracked edits carry no " + "author, so the check covers them whoever made them — the name marks " + "the run as redlining work and is not used to filter. Requires " + "--original; docx only.", + ) + args = parser.parse_args() + + if args.author is not None and not args.original: + _fail("--author requires --original") + + path = Path(args.path) + if not path.exists(): + _fail(f"{path} does not exist") + + original_file = None + if args.original: + original_file = Path(args.original) + if not original_file.is_file(): + _fail(f"{original_file} is not a file") + if original_file.suffix.lower() not in OOXML_FAMILY: + _fail(f"{original_file} must be one of: {', '.join(sorted(OOXML_FAMILY))}") + + family = OOXML_FAMILY.get((original_file or path).suffix.lower()) + if family is None: + _fail( + f"Cannot determine file type from {path}. Use --original or provide one of: {', '.join(sorted(OOXML_FAMILY))}." + ) + + if args.author is not None and family != "docx": + _fail(f"--author only applies to docx files, not {family}") + + packed_file = None + temp_dir_ctx = None + if path.is_file() and path.suffix.lower() in OOXML_FAMILY: + packed_file = path + temp_dir_ctx = tempfile.TemporaryDirectory() + unpacked_dir = Path(temp_dir_ctx.name) + try: + with zipfile.ZipFile(path, "r") as zf: + safe_extract(zf, unpacked_dir) + except (zipfile.BadZipFile, ValueError, OSError) as e: + _fail(f"cannot unpack {path}: {e}") + else: + if not path.is_dir(): + _fail(f"{path} is not a directory or Office file") + unpacked_dir = path + + match family: + case "docx": + validators = [ + DOCXSchemaValidator(unpacked_dir, original_file, verbose=args.verbose), + ] + if args.author is not None: + validators.append( + RedliningValidator(unpacked_dir, original_file, verbose=args.verbose) + ) + elif original_file and _has_tracked_changes(unpacked_dir): + print( + "Note: this document has tracked changes; they were not " + "checked against the original (pass --author to check)." + ) + case "pptx": + validators = [ + PPTXSchemaValidator(unpacked_dir, original_file, verbose=args.verbose), + ] + case "xlsx": + exts = ", ".join(k for k, v in sorted(OOXML_FAMILY.items()) if v == "xlsx") + print( + f"No XSD schema validation is performed for xlsx-family files ({exts}). " + "For formula-error checking, use scripts/recalc.py instead." + ) + sys.exit(0) + case _: + print(f"Error: Validation not supported for file type {family}") + sys.exit(1) + + if args.auto_repair: + total_repairs = sum(v.repair() for v in validators) + if total_repairs: + print(f"Auto-repaired {total_repairs} issue(s)") + if packed_file is not None: + rezip(unpacked_dir, packed_file) + print(f"Wrote repaired file to {packed_file}") + + success = all([v.validate() for v in validators]) + + if temp_dir_ctx is not None: + temp_dir_ctx.cleanup() + + if success: + print("All validations PASSED!") + + sys.exit(0 if success else 1) + + +if __name__ == "__main__": + main() diff --git a/.github/skills/anthropic-docx/scripts/office/validators/__init__.py b/.github/skills/anthropic-docx/scripts/office/validators/__init__.py new file mode 100644 index 00000000..db092ece --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/validators/__init__.py @@ -0,0 +1,15 @@ +""" +Validation modules for Word document processing. +""" + +from .base import BaseSchemaValidator +from .docx import DOCXSchemaValidator +from .pptx import PPTXSchemaValidator +from .redlining import RedliningValidator + +__all__ = [ + "BaseSchemaValidator", + "DOCXSchemaValidator", + "PPTXSchemaValidator", + "RedliningValidator", +] diff --git a/.github/skills/anthropic-docx/scripts/office/validators/base.py b/.github/skills/anthropic-docx/scripts/office/validators/base.py new file mode 100644 index 00000000..33fc97bb --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/validators/base.py @@ -0,0 +1,875 @@ +""" +Base validator with common validation logic for document files. +""" + +import re +from pathlib import Path + +import defusedxml.minidom +from functools import lru_cache + +import lxml.etree + +from helpers import safe_extract + + +@lru_cache(maxsize=None) +def _load_schema(schema_path: str): + with open(schema_path, "rb") as xsd_file: + xsd_doc = lxml.etree.parse( + xsd_file, parser=lxml.etree.XMLParser(), base_url=schema_path + ) + return lxml.etree.XMLSchema(xsd_doc) + +class BaseSchemaValidator: + + IGNORED_VALIDATION_ERRORS = [ + "hyphenationZone", + "purl.org/dc/terms", + ] + + UNIQUE_ID_REQUIREMENTS = { + "comment": ("id", "file"), + "commentrangestart": ("id", "file"), + "commentrangeend": ("id", "file"), + "bookmarkstart": ("id", "file"), + "bookmarkend": ("id", "file"), + "sldid": ("id", "file"), + "sldmasterid": ("id", "global"), + "sldlayoutid": ("id", "global"), + "cm": ("authorid", "file"), + "sheet": ("sheetid", "file"), + "definedname": ("id", "file"), + "cxnsp": ("id", "file"), + "sp": ("id", "file"), + "pic": ("id", "file"), + "grpsp": ("id", "file"), + } + + EXCLUDED_ID_CONTAINERS = { + "sectionlst", + } + + ELEMENT_RELATIONSHIP_TYPES = {} + + SCHEMA_MAPPINGS = { + "word": "ISO-IEC29500-4_2016/wml.xsd", + "ppt": "ISO-IEC29500-4_2016/pml.xsd", + "xl": "ISO-IEC29500-4_2016/sml.xsd", + "[Content_Types].xml": "ecma/fouth-edition/opc-contentTypes.xsd", + "app.xml": "ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd", + "core.xml": "ecma/fouth-edition/opc-coreProperties.xsd", + "custom.xml": "ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd", + ".rels": "ecma/fouth-edition/opc-relationships.xsd", + "people.xml": "microsoft/wml-2012.xsd", + "commentsIds.xml": "microsoft/wml-cid-2016.xsd", + "commentsExtensible.xml": "microsoft/wml-cex-2018.xsd", + "commentsExtended.xml": "microsoft/wml-2012.xsd", + "chart": "ISO-IEC29500-4_2016/dml-chart.xsd", + "theme": "ISO-IEC29500-4_2016/dml-main.xsd", + "drawing": "ISO-IEC29500-4_2016/dml-main.xsd", + } + + MC_NAMESPACE = "http://schemas.openxmlformats.org/markup-compatibility/2006" + XML_NAMESPACE = "http://www.w3.org/XML/1998/namespace" + + PACKAGE_RELATIONSHIPS_NAMESPACE = ( + "http://schemas.openxmlformats.org/package/2006/relationships" + ) + OFFICE_RELATIONSHIPS_NAMESPACE = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships" + ) + CONTENT_TYPES_NAMESPACE = ( + "http://schemas.openxmlformats.org/package/2006/content-types" + ) + + MAIN_CONTENT_FOLDERS = {"word", "ppt", "xl"} + + OOXML_NAMESPACES = { + "http://schemas.openxmlformats.org/officeDocument/2006/math", + "http://schemas.openxmlformats.org/officeDocument/2006/relationships", + "http://schemas.openxmlformats.org/schemaLibrary/2006/main", + "http://schemas.openxmlformats.org/drawingml/2006/main", + "http://schemas.openxmlformats.org/drawingml/2006/chart", + "http://schemas.openxmlformats.org/drawingml/2006/chartDrawing", + "http://schemas.openxmlformats.org/drawingml/2006/diagram", + "http://schemas.openxmlformats.org/drawingml/2006/picture", + "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing", + "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing", + "http://schemas.openxmlformats.org/wordprocessingml/2006/main", + "http://schemas.openxmlformats.org/presentationml/2006/main", + "http://schemas.openxmlformats.org/spreadsheetml/2006/main", + "http://schemas.openxmlformats.org/officeDocument/2006/sharedTypes", + "http://www.w3.org/XML/1998/namespace", + } + + def __init__(self, unpacked_dir, original_file=None, verbose=False): + self.unpacked_dir = Path(unpacked_dir).resolve() + self.original_file = Path(original_file) if original_file else None + self.verbose = verbose + + self.schemas_dir = Path(__file__).parent.parent / "schemas" + + patterns = ["*.xml", "*.rels"] + self.xml_files = [ + f for pattern in patterns for f in self.unpacked_dir.rglob(pattern) + ] + + if not self.xml_files: + print(f"Warning: No XML files found in {self.unpacked_dir}") + + def validate(self): + raise NotImplementedError("Subclasses must implement the validate method") + + def repair(self) -> int: + return self.repair_whitespace_preservation() + + def repair_whitespace_preservation(self) -> int: + repairs = 0 + + for xml_file in self.xml_files: + try: + content = xml_file.read_text(encoding="utf-8") + dom = defusedxml.minidom.parseString(content) + pending = [] + + for elem in dom.getElementsByTagName("*"): + local_name = elem.tagName.rsplit(":", 1)[-1] + if local_name in ("t", "delText", "instrText", "delInstrText"): + text = "".join( + child.data + for child in elem.childNodes + if child.nodeType in (child.TEXT_NODE, child.CDATA_SECTION_NODE) + ) + ws = (" ", "\t", "\n", "\r") + if text and (text.startswith(ws) or text.endswith(ws)): + if elem.getAttribute("xml:space") != "preserve": + elem.setAttribute("xml:space", "preserve") + text_preview = repr(text[:30]) + "..." if len(text) > 30 else repr(text) + pending.append(f" Repaired: {xml_file.name}: Added xml:space='preserve' to {elem.tagName}: {text_preview}") + + if pending: + xml_file.write_bytes(dom.toxml(encoding="UTF-8")) + for message in pending: + print(message) + repairs += len(pending) + + except Exception: + pass + + return repairs + + def validate_xml(self): + errors = [] + + for xml_file in self.xml_files: + try: + lxml.etree.parse(str(xml_file)) + except lxml.etree.XMLSyntaxError as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {e.lineno}: {e.msg}" + ) + except Exception as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Unexpected error: {str(e)}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} XML violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All XML files are well-formed") + return True + + def validate_namespaces(self): + errors = [] + + for xml_file in self.xml_files: + try: + root = lxml.etree.parse(str(xml_file)).getroot() + declared = set(root.nsmap.keys()) - {None} + + for attr_val in [ + v for k, v in root.attrib.items() if k.endswith("Ignorable") + ]: + undeclared = set(attr_val.split()) - declared + errors.extend( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Namespace '{ns}' in Ignorable but not declared" + for ns in undeclared + ) + except lxml.etree.XMLSyntaxError: + continue + + if errors: + print(f"FAILED - {len(errors)} namespace issues:") + for error in errors: + print(error) + return False + if self.verbose: + print("PASSED - All namespace prefixes properly declared") + return True + + def validate_unique_ids(self): + errors = [] + global_ids = {} + + for xml_file in self.xml_files: + try: + root = lxml.etree.parse(str(xml_file)).getroot() + file_ids = {} + + mc_elements = root.xpath( + ".//mc:AlternateContent", namespaces={"mc": self.MC_NAMESPACE} + ) + for elem in mc_elements: + elem.getparent().remove(elem) + + for elem in root.iter(): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + tag = ( + elem.tag.split("}")[-1].lower() + if "}" in elem.tag + else elem.tag.lower() + ) + + if tag in self.UNIQUE_ID_REQUIREMENTS: + in_excluded_container = any( + ancestor.tag.split("}")[-1].lower() in self.EXCLUDED_ID_CONTAINERS + for ancestor in elem.iterancestors() + ) + if in_excluded_container: + continue + + attr_name, scope = self.UNIQUE_ID_REQUIREMENTS[tag] + + id_value = None + for attr, value in elem.attrib.items(): + attr_local = ( + attr.split("}")[-1].lower() + if "}" in attr + else attr.lower() + ) + if attr_local == attr_name: + id_value = value + break + + if id_value is not None: + if scope == "global": + if id_value in global_ids: + prev_file, prev_line, prev_tag = global_ids[ + id_value + ] + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: Global ID '{id_value}' in <{tag}> " + f"already used in {prev_file} at line {prev_line} in <{prev_tag}>" + ) + else: + global_ids[id_value] = ( + xml_file.relative_to(self.unpacked_dir), + elem.sourceline, + tag, + ) + elif scope == "file": + key = (tag, attr_name) + if key not in file_ids: + file_ids[key] = {} + + if id_value in file_ids[key]: + prev_line = file_ids[key][id_value] + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: Duplicate {attr_name}='{id_value}' in <{tag}> " + f"(first occurrence at line {prev_line})" + ) + else: + file_ids[key][id_value] = elem.sourceline + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} ID uniqueness violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All required IDs are unique") + return True + + def validate_file_references(self): + errors = [] + + rels_files = list(self.unpacked_dir.rglob("*.rels")) + + if not rels_files: + if self.verbose: + print("PASSED - No .rels files found") + return True + + all_files = [] + for file_path in self.unpacked_dir.rglob("*"): + if ( + file_path.is_file() + and file_path.name != "[Content_Types].xml" + and not file_path.name.endswith(".rels") + ): + all_files.append(file_path.resolve()) + + all_referenced_files = set() + + if self.verbose: + print( + f"Found {len(rels_files)} .rels files and {len(all_files)} target files" + ) + + for rels_file in rels_files: + try: + rels_root = lxml.etree.parse(str(rels_file)).getroot() + + rels_dir = rels_file.parent + + referenced_files = set() + broken_refs = [] + + for rel in rels_root.findall( + ".//ns:Relationship", + namespaces={"ns": self.PACKAGE_RELATIONSHIPS_NAMESPACE}, + ): + target = rel.get("Target") + if rel.get("TargetMode") == "External": + continue + if target and not target.startswith( + ("http", "mailto:") + ): + if target.startswith("/"): + target_path = self.unpacked_dir / target.lstrip("/") + elif rels_file.name == ".rels": + target_path = self.unpacked_dir / target + else: + base_dir = rels_dir.parent + target_path = base_dir / target + + try: + target_path = target_path.resolve() + if target_path.exists() and target_path.is_file(): + referenced_files.add(target_path) + all_referenced_files.add(target_path) + else: + broken_refs.append((target, rel.sourceline)) + except (OSError, ValueError): + broken_refs.append((target, rel.sourceline)) + + if broken_refs: + rel_path = rels_file.relative_to(self.unpacked_dir) + for broken_ref, line_num in broken_refs: + errors.append( + f" {rel_path}: Line {line_num}: Broken reference to {broken_ref}" + ) + + except Exception as e: + rel_path = rels_file.relative_to(self.unpacked_dir) + errors.append(f" Error parsing {rel_path}: {e}") + + unreferenced_files = set(all_files) - all_referenced_files + + if unreferenced_files: + for unref_file in sorted(unreferenced_files): + unref_rel_path = unref_file.relative_to(self.unpacked_dir) + errors.append(f" Unreferenced file: {unref_rel_path}") + + if errors: + print(f"FAILED - Found {len(errors)} relationship validation errors:") + for error in errors: + print(error) + print( + "CRITICAL: These errors will cause the document to appear corrupt. " + + "Broken references MUST be fixed, " + + "and unreferenced files MUST be referenced or removed." + ) + return False + else: + if self.verbose: + print( + "PASSED - All references are valid and all files are properly referenced" + ) + return True + + def validate_all_relationship_ids(self): + import lxml.etree + + errors = [] + + for xml_file in self.xml_files: + if xml_file.suffix == ".rels": + continue + + rels_dir = xml_file.parent / "_rels" + rels_file = rels_dir / f"{xml_file.name}.rels" + + if not rels_file.exists(): + continue + + try: + rels_root = lxml.etree.parse(str(rels_file)).getroot() + rid_to_type = {} + + for rel in rels_root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ): + rid = rel.get("Id") + rel_type = rel.get("Type", "") + if rid: + if rid in rid_to_type: + rels_rel_path = rels_file.relative_to(self.unpacked_dir) + errors.append( + f" {rels_rel_path}: Line {rel.sourceline}: " + f"Duplicate relationship ID '{rid}' (IDs must be unique)" + ) + type_name = ( + rel_type.split("/")[-1] if "/" in rel_type else rel_type + ) + rid_to_type[rid] = type_name + + xml_root = lxml.etree.parse(str(xml_file)).getroot() + + r_ns = self.OFFICE_RELATIONSHIPS_NAMESPACE + rid_attrs_to_check = ["id", "embed", "link"] + for elem in xml_root.iter(): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + for attr_name in rid_attrs_to_check: + rid_attr = elem.get(f"{{{r_ns}}}{attr_name}") + if not rid_attr: + continue + xml_rel_path = xml_file.relative_to(self.unpacked_dir) + elem_name = ( + elem.tag.split("}")[-1] if "}" in elem.tag else elem.tag + ) + + if rid_attr not in rid_to_type: + errors.append( + f" {xml_rel_path}: Line {elem.sourceline}: " + f"<{elem_name}> r:{attr_name} references non-existent relationship '{rid_attr}' " + f"(valid IDs: {', '.join(sorted(rid_to_type.keys())[:5])}{'...' if len(rid_to_type) > 5 else ''})" + ) + elif attr_name == "id" and self.ELEMENT_RELATIONSHIP_TYPES: + expected_type = self._get_expected_relationship_type( + elem_name + ) + if expected_type: + actual_type = rid_to_type[rid_attr] + if expected_type not in actual_type.lower(): + errors.append( + f" {xml_rel_path}: Line {elem.sourceline}: " + f"<{elem_name}> references '{rid_attr}' which points to '{actual_type}' " + f"but should point to a '{expected_type}' relationship" + ) + + except Exception as e: + xml_rel_path = xml_file.relative_to(self.unpacked_dir) + errors.append(f" Error processing {xml_rel_path}: {e}") + + if errors: + print(f"FAILED - Found {len(errors)} relationship ID reference errors:") + for error in errors: + print(error) + print("\nThese ID mismatches will cause the document to appear corrupt!") + return False + else: + if self.verbose: + print("PASSED - All relationship ID references are valid") + return True + + def _get_expected_relationship_type(self, element_name): + elem_lower = element_name.lower() + + if elem_lower in self.ELEMENT_RELATIONSHIP_TYPES: + return self.ELEMENT_RELATIONSHIP_TYPES[elem_lower] + + if elem_lower.endswith("id") and len(elem_lower) > 2: + prefix = elem_lower[:-2] + if prefix.endswith("master"): + return prefix.lower() + elif prefix.endswith("layout"): + return prefix.lower() + else: + if prefix == "sld": + return "slide" + return prefix.lower() + + if elem_lower.endswith("reference") and len(elem_lower) > 9: + prefix = elem_lower[:-9] + return prefix.lower() + + return None + + def validate_content_types(self): + errors = [] + + content_types_file = self.unpacked_dir / "[Content_Types].xml" + if not content_types_file.exists(): + print("FAILED - [Content_Types].xml file not found") + return False + + try: + root = lxml.etree.parse(str(content_types_file)).getroot() + declared_parts = set() + declared_extensions = set() + + for override in root.findall( + f".//{{{self.CONTENT_TYPES_NAMESPACE}}}Override" + ): + part_name = override.get("PartName") + if part_name is not None: + declared_parts.add(part_name.lstrip("/")) + + for default in root.findall( + f".//{{{self.CONTENT_TYPES_NAMESPACE}}}Default" + ): + extension = default.get("Extension") + if extension is not None: + declared_extensions.add(extension.lower()) + + declarable_roots = { + "sld", + "sldLayout", + "sldMaster", + "presentation", + "document", + "workbook", + "worksheet", + "theme", + } + + media_extensions = { + "png": "image/png", + "jpg": "image/jpeg", + "jpeg": "image/jpeg", + "gif": "image/gif", + "bmp": "image/bmp", + "tiff": "image/tiff", + "wmf": "image/x-wmf", + "emf": "image/x-emf", + } + + all_files = list(self.unpacked_dir.rglob("*")) + all_files = [f for f in all_files if f.is_file()] + + for xml_file in self.xml_files: + path_str = str(xml_file.relative_to(self.unpacked_dir)).replace( + "\\", "/" + ) + + if any( + skip in path_str + for skip in [".rels", "[Content_Types]", "docProps/", "_rels/"] + ): + continue + + try: + root_tag = lxml.etree.parse(str(xml_file)).getroot().tag + root_name = root_tag.split("}")[-1] if "}" in root_tag else root_tag + + if root_name in declarable_roots and path_str not in declared_parts: + errors.append( + f" {path_str}: File with <{root_name}> root not declared in [Content_Types].xml" + ) + + except Exception: + continue + + for file_path in all_files: + if file_path.suffix.lower() in {".xml", ".rels"}: + continue + if file_path.name == "[Content_Types].xml": + continue + if "_rels" in file_path.parts or "docProps" in file_path.parts: + continue + + extension = file_path.suffix.lstrip(".").lower() + if extension and extension not in declared_extensions: + if extension in media_extensions: + relative_path = file_path.relative_to(self.unpacked_dir) + errors.append( + f' {relative_path}: File with extension \'{extension}\' not declared in [Content_Types].xml - should add: ' + ) + + except Exception as e: + errors.append(f" Error parsing [Content_Types].xml: {e}") + + if errors: + print(f"FAILED - Found {len(errors)} content type declaration errors:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print( + "PASSED - All content files are properly declared in [Content_Types].xml" + ) + return True + + def validate_file_against_xsd(self, xml_file, verbose=False): + xml_file = Path(xml_file).resolve() + unpacked_dir = self.unpacked_dir.resolve() + + is_valid, current_errors = self._validate_single_file_xsd( + xml_file, unpacked_dir + ) + + if is_valid is None: + return None, set() + elif is_valid: + return True, set() + + original_errors = self._get_original_file_errors(xml_file) + + assert current_errors is not None + new_errors = current_errors - original_errors + + new_errors = { + e for e in new_errors + if not any(pattern in e for pattern in self.IGNORED_VALIDATION_ERRORS) + } + + if new_errors: + if verbose: + relative_path = xml_file.relative_to(unpacked_dir) + print(f"FAILED - {relative_path}: {len(new_errors)} new error(s)") + for error in list(new_errors)[:3]: + truncated = error[:250] + "..." if len(error) > 250 else error + print(f" - {truncated}") + return False, new_errors + else: + if verbose: + print( + f"PASSED - No new errors (original had {len(current_errors)} errors)" + ) + return True, set() + + def validate_against_xsd(self): + new_errors = [] + original_error_count = 0 + valid_count = 0 + skipped_count = 0 + + for xml_file in self.xml_files: + relative_path = str(xml_file.relative_to(self.unpacked_dir)) + is_valid, new_file_errors = self.validate_file_against_xsd( + xml_file, verbose=False + ) + + if is_valid is None: + skipped_count += 1 + continue + elif is_valid and not new_file_errors: + valid_count += 1 + continue + elif is_valid: + original_error_count += 1 + valid_count += 1 + continue + + new_errors.append(f" {relative_path}: {len(new_file_errors)} new error(s)") + for error in list(new_file_errors)[:3]: + new_errors.append( + f" - {error[:250]}..." if len(error) > 250 else f" - {error}" + ) + + if self.verbose: + print(f"Validated {len(self.xml_files)} files:") + print(f" - Valid: {valid_count}") + print(f" - Skipped (no schema): {skipped_count}") + if original_error_count: + print(f" - With original errors (ignored): {original_error_count}") + print( + f" - With NEW errors: {len(new_errors) > 0 and len([e for e in new_errors if not e.startswith(' ')]) or 0}" + ) + + if new_errors: + print("\nFAILED - Found NEW validation errors:") + for error in new_errors: + print(error) + return False + else: + if self.verbose: + print("\nPASSED - No new XSD validation errors introduced") + return True + + def _get_schema_path(self, xml_file): + if xml_file.name in self.SCHEMA_MAPPINGS: + return self.schemas_dir / self.SCHEMA_MAPPINGS[xml_file.name] + + if xml_file.suffix == ".rels": + return self.schemas_dir / self.SCHEMA_MAPPINGS[".rels"] + + if "charts/" in str(xml_file) and xml_file.name.startswith("chart"): + return self.schemas_dir / self.SCHEMA_MAPPINGS["chart"] + + if "theme/" in str(xml_file) and xml_file.name.startswith("theme"): + return self.schemas_dir / self.SCHEMA_MAPPINGS["theme"] + + if xml_file.parent.name in self.MAIN_CONTENT_FOLDERS: + return self.schemas_dir / self.SCHEMA_MAPPINGS[xml_file.parent.name] + + return None + + def _clean_ignorable_namespaces(self, xml_doc): + xml_string = lxml.etree.tostring(xml_doc, encoding="unicode") + xml_copy = lxml.etree.fromstring(xml_string) + + for elem in xml_copy.iter(): + attrs_to_remove = [] + + for attr in elem.attrib: + if "{" in attr: + ns = attr.split("}")[0][1:] + if ns not in self.OOXML_NAMESPACES: + attrs_to_remove.append(attr) + + for attr in attrs_to_remove: + del elem.attrib[attr] + + self._remove_ignorable_elements(xml_copy) + + return lxml.etree.ElementTree(xml_copy) + + def _remove_ignorable_elements(self, root): + elements_to_remove = [] + + for elem in list(root): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + + tag_str = str(elem.tag) + if tag_str.startswith("{"): + ns = tag_str.split("}")[0][1:] + if ns not in self.OOXML_NAMESPACES: + elements_to_remove.append(elem) + continue + + self._remove_ignorable_elements(elem) + + for elem in elements_to_remove: + root.remove(elem) + + def _preprocess_for_mc_ignorable(self, xml_doc): + root = xml_doc.getroot() + + if f"{{{self.MC_NAMESPACE}}}Ignorable" in root.attrib: + del root.attrib[f"{{{self.MC_NAMESPACE}}}Ignorable"] + + return xml_doc + + def _preprocess_for_schema(self, xml_doc, relative_path): + return xml_doc + + def _validate_single_file_xsd(self, xml_file, base_path, schema_path=None): + schema_path = schema_path or self._get_schema_path(xml_file) + if not schema_path: + return None, None + + try: + schema = _load_schema(str(schema_path)) + + with open(xml_file, "r") as f: + xml_doc = lxml.etree.parse(f) + + xml_doc, _ = self._remove_template_tags_from_text_nodes(xml_doc) + xml_doc = self._preprocess_for_mc_ignorable(xml_doc) + + relative_path = xml_file.relative_to(base_path) + if ( + relative_path.parts + and relative_path.parts[0] in self.MAIN_CONTENT_FOLDERS + ): + xml_doc = self._clean_ignorable_namespaces(xml_doc) + + xml_doc = self._preprocess_for_schema(xml_doc, relative_path) + + if schema.validate(xml_doc): + return True, set() + else: + errors = set() + for error in schema.error_log: + errors.add(error.message) + return False, errors + + except Exception as e: + return False, {str(e)} + + def _get_original_file_errors(self, xml_file, schema_path=None): + if self.original_file is None: + return set() + + import tempfile + import zipfile + + xml_file = Path(xml_file).resolve() + unpacked_dir = self.unpacked_dir.resolve() + relative_path = xml_file.relative_to(unpacked_dir) + + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + + try: + with zipfile.ZipFile(self.original_file, "r") as zip_ref: + safe_extract(zip_ref, temp_path) + except (zipfile.BadZipFile, ValueError, OSError): + return set() + + original_xml_file = temp_path / relative_path + + if not original_xml_file.exists(): + return set() + + is_valid, errors = self._validate_single_file_xsd( + original_xml_file, temp_path, schema_path=schema_path + ) + return errors if errors else set() + + def _remove_template_tags_from_text_nodes(self, xml_doc): + warnings = [] + template_pattern = re.compile(r"\{\{[^}]*\}\}") + + xml_string = lxml.etree.tostring(xml_doc, encoding="unicode") + xml_copy = lxml.etree.fromstring(xml_string) + + def process_text_content(text, content_type): + if not text: + return text + matches = list(template_pattern.finditer(text)) + if matches: + for match in matches: + warnings.append( + f"Found template tag in {content_type}: {match.group()}" + ) + return template_pattern.sub("", text) + return text + + for elem in xml_copy.iter(): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + tag_str = str(elem.tag) + if tag_str.endswith("}t") or tag_str == "t": + continue + + elem.text = process_text_content(elem.text, "text content") + elem.tail = process_text_content(elem.tail, "tail content") + + return lxml.etree.ElementTree(xml_copy), warnings + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-docx/scripts/office/validators/docx.py b/.github/skills/anthropic-docx/scripts/office/validators/docx.py new file mode 100644 index 00000000..b1814994 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/validators/docx.py @@ -0,0 +1,466 @@ +""" +Validator for Word document XML files against XSD schemas. +""" + +import random +import re +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.minidom +import lxml.etree + +from helpers import safe_extract + +from .base import BaseSchemaValidator + + +class DOCXSchemaValidator(BaseSchemaValidator): + + WORD_2006_NAMESPACE = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + W14_NAMESPACE = "http://schemas.microsoft.com/office/word/2010/wordml" + W16CID_NAMESPACE = "http://schemas.microsoft.com/office/word/2016/wordml/cid" + + ELEMENT_RELATIONSHIP_TYPES = {} + + def validate(self): + if not self.validate_xml(): + return False + + all_valid = True + if not self.validate_namespaces(): + all_valid = False + + if not self.validate_unique_ids(): + all_valid = False + + if not self.validate_file_references(): + all_valid = False + + if not self.validate_content_types(): + all_valid = False + + if not self.validate_against_xsd(): + all_valid = False + + if not self.validate_whitespace_preservation(): + all_valid = False + + if not self.validate_deletions(): + all_valid = False + + if not self.validate_insertions(): + all_valid = False + + if not self.validate_all_relationship_ids(): + all_valid = False + + if not self.validate_id_constraints(): + all_valid = False + + if not self.validate_comment_markers(): + all_valid = False + + self.compare_paragraph_counts() + + return all_valid + + def validate_whitespace_preservation(self): + errors = [] + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + + for elem in root.iter(f"{{{self.WORD_2006_NAMESPACE}}}t"): + if elem.text: + text = elem.text + if re.search(r"^[ \t\n\r]", text) or re.search( + r"[ \t\n\r]$", text + ): + xml_space_attr = f"{{{self.XML_NAMESPACE}}}space" + if ( + xml_space_attr not in elem.attrib + or elem.attrib[xml_space_attr] != "preserve" + ): + text_preview = ( + repr(text)[:50] + "..." + if len(repr(text)) > 50 + else repr(text) + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: w:t element with whitespace missing xml:space='preserve': {text_preview}" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} whitespace preservation violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All whitespace is properly preserved") + return True + + def validate_deletions(self): + errors = [] + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + namespaces = {"w": self.WORD_2006_NAMESPACE} + + for t_elem in root.xpath(".//w:del//w:t", namespaces=namespaces): + if t_elem.text: + text_preview = ( + repr(t_elem.text)[:50] + "..." + if len(repr(t_elem.text)) > 50 + else repr(t_elem.text) + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {t_elem.sourceline}: found within : {text_preview}" + ) + + for instr_elem in root.xpath( + ".//w:del//w:instrText", namespaces=namespaces + ): + text_preview = ( + repr(instr_elem.text or "")[:50] + "..." + if len(repr(instr_elem.text or "")) > 50 + else repr(instr_elem.text or "") + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {instr_elem.sourceline}: found within (use ): {text_preview}" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} deletion validation violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - No w:t elements found within w:del elements") + return True + + def count_paragraphs_in_unpacked(self): + count = 0 + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + paragraphs = root.findall(f".//{{{self.WORD_2006_NAMESPACE}}}p") + count = len(paragraphs) + except Exception as e: + print(f"Error counting paragraphs in unpacked document: {e}") + + return count + + def count_paragraphs_in_original(self): + original = self.original_file + if original is None: + return 0 + + count = 0 + + try: + with tempfile.TemporaryDirectory() as temp_dir: + with zipfile.ZipFile(original, "r") as zip_ref: + safe_extract(zip_ref, Path(temp_dir)) + + doc_xml_path = temp_dir + "/word/document.xml" + root = lxml.etree.parse(doc_xml_path).getroot() + + paragraphs = root.findall(f".//{{{self.WORD_2006_NAMESPACE}}}p") + count = len(paragraphs) + + except Exception as e: + print(f"Error counting paragraphs in original document: {e}") + + return count + + def validate_insertions(self): + errors = [] + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + namespaces = {"w": self.WORD_2006_NAMESPACE} + + invalid_elements = root.xpath( + ".//w:ins//w:delText[not(ancestor::w:del)]", namespaces=namespaces + ) + + for elem in invalid_elements: + text_preview = ( + repr(elem.text or "")[:50] + "..." + if len(repr(elem.text or "")) > 50 + else repr(elem.text or "") + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: within : {text_preview}" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} insertion validation violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - No w:delText elements within w:ins elements") + return True + + def compare_paragraph_counts(self): + new_count = self.count_paragraphs_in_unpacked() + if self.original_file is None: + print(f"\nParagraphs: {new_count}") + return + + original_count = self.count_paragraphs_in_original() + diff = new_count - original_count + diff_str = f"+{diff}" if diff > 0 else str(diff) + print(f"\nParagraphs: {original_count} → {new_count} ({diff_str})") + + def _parse_id_value(self, val: str, base: int = 16) -> int: + return int(val, base) + + def validate_id_constraints(self): + errors = [] + para_id_attr = f"{{{self.W14_NAMESPACE}}}paraId" + durable_id_attr = f"{{{self.W16CID_NAMESPACE}}}durableId" + + for xml_file in self.xml_files: + try: + for elem in lxml.etree.parse(str(xml_file)).iter(): + if val := elem.get(para_id_attr): + try: + if self._parse_id_value(val, base=16) >= 0x80000000: + errors.append( + f" {xml_file.name}:{elem.sourceline}: paraId={val} >= 0x80000000" + ) + except ValueError: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"paraId={val} is not valid hex" + ) + + if val := elem.get(durable_id_attr): + if xml_file.name == "numbering.xml": + try: + if self._parse_id_value(val, base=10) >= 0x7FFFFFFF: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} >= 0x7FFFFFFF" + ) + except ValueError: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} must be decimal in numbering.xml" + ) + else: + try: + if self._parse_id_value(val, base=16) >= 0x7FFFFFFF: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} >= 0x7FFFFFFF" + ) + except ValueError: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} is not valid hex" + ) + except lxml.etree.XMLSyntaxError: + continue + + if errors: + print(f"FAILED - {len(errors)} ID constraint violations:") + for e in errors: + print(e) + elif self.verbose: + print("PASSED - All paraId/durableId values within constraints") + return not errors + + def validate_comment_markers(self): + errors = [] + + document_xml = None + comments_xml = None + for xml_file in self.xml_files: + if xml_file.name == "document.xml" and "word" in str(xml_file): + document_xml = xml_file + elif xml_file.name == "comments.xml": + comments_xml = xml_file + + if not document_xml: + if self.verbose: + print("PASSED - No document.xml found (skipping comment validation)") + return True + + try: + doc_root = lxml.etree.parse(str(document_xml)).getroot() + namespaces = {"w": self.WORD_2006_NAMESPACE} + + range_starts = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in doc_root.xpath( + ".//w:commentRangeStart", namespaces=namespaces + ) + } + range_ends = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in doc_root.xpath( + ".//w:commentRangeEnd", namespaces=namespaces + ) + } + references = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in doc_root.xpath( + ".//w:commentReference", namespaces=namespaces + ) + } + + orphaned_ends = range_ends - range_starts + for comment_id in sorted( + orphaned_ends, key=lambda x: int(x) if x and x.isdigit() else 0 + ): + errors.append( + f' document.xml: commentRangeEnd id="{comment_id}" has no matching commentRangeStart' + ) + + orphaned_starts = range_starts - range_ends + for comment_id in sorted( + orphaned_starts, key=lambda x: int(x) if x and x.isdigit() else 0 + ): + errors.append( + f' document.xml: commentRangeStart id="{comment_id}" has no matching commentRangeEnd' + ) + + comment_ids = set() + if comments_xml and comments_xml.exists(): + comments_root = lxml.etree.parse(str(comments_xml)).getroot() + comment_ids = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in comments_root.xpath( + ".//w:comment", namespaces=namespaces + ) + } + + marker_ids = range_starts | range_ends | references + invalid_refs = marker_ids - comment_ids + for comment_id in sorted( + invalid_refs, key=lambda x: int(x) if x and x.isdigit() else 0 + ): + if comment_id: + errors.append( + f' document.xml: marker id="{comment_id}" references non-existent comment' + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append(f" Error parsing XML: {e}") + + if errors: + print(f"FAILED - {len(errors)} comment marker violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All comment markers properly paired") + return True + + def repair(self) -> int: + repairs = super().repair() + repairs += self.repair_durableId() + return repairs + + def repair_durableId(self) -> int: + DURABLE_ID_ATTRS = ("w16cid:durableId", "w16cex:durableId") + repairs = 0 + renames: dict = {} + + for xml_file in self.xml_files: + try: + content = xml_file.read_text(encoding="utf-8") + dom = defusedxml.minidom.parseString(content) + is_numbering = xml_file.name == "numbering.xml" + base = 10 if is_numbering else 16 + pending = [] + seen_in_file = set() + modified = False + + for elem in dom.getElementsByTagName("*"): + for attr_name in DURABLE_ID_ATTRS: + if not elem.hasAttribute(attr_name): + continue + + durable_id = elem.getAttribute(attr_name) + try: + key = self._parse_id_value(durable_id, base=base) + needs_repair = key >= 0x7FFFFFFF + except ValueError: + key = durable_id + needs_repair = True + + if needs_repair: + if key in seen_in_file: + value = random.randint(1, 0x7FFFFFFE) + else: + seen_in_file.add(key) + if key not in renames: + renames[key] = random.randint(1, 0x7FFFFFFE) + value = renames[key] + new_id = str(value) if is_numbering else f"{value:08X}" + + elem.setAttribute(attr_name, new_id) + pending.append( + f" Repaired: {xml_file.name}: durableId {durable_id} → {new_id}" + ) + modified = True + + if modified: + xml_file.write_bytes(dom.toxml(encoding="UTF-8")) + for message in pending: + print(message) + repairs += len(pending) + + except Exception: + pass + + return repairs + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-docx/scripts/office/validators/pptx.py b/.github/skills/anthropic-docx/scripts/office/validators/pptx.py new file mode 100644 index 00000000..318f0e61 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/validators/pptx.py @@ -0,0 +1,441 @@ +""" +Validator for PowerPoint presentation XML files against XSD schemas. +""" + +import re +from pathlib import Path + +from helpers import opc_target, rels_source_part, safe_extract + +from .base import BaseSchemaValidator + + +class PPTXSchemaValidator(BaseSchemaValidator): + + PRESENTATIONML_NAMESPACE = ( + "http://schemas.openxmlformats.org/presentationml/2006/main" + ) + + ELEMENT_RELATIONSHIP_TYPES = { + "sldid": "slide", + "sldmasterid": "slidemaster", + "notesmasterid": "notesmaster", + "sldlayoutid": "slidelayout", + "themeid": "theme", + "tablestyleid": "tablestyles", + } + + def validate(self): + if not self.validate_xml(): + return False + + all_valid = True + if not self.validate_namespaces(): + all_valid = False + + if not self.validate_unique_ids(): + all_valid = False + + if not self.validate_uuid_ids(): + all_valid = False + + if not self.validate_file_references(): + all_valid = False + + if not self.validate_slide_layout_ids(): + all_valid = False + + if not self.validate_content_types(): + all_valid = False + + if not self.validate_against_xsd(): + all_valid = False + + if not self.validate_notes_slide_references(): + all_valid = False + + if not self.validate_all_relationship_ids(): + all_valid = False + + if not self.validate_no_duplicate_slide_layouts(): + all_valid = False + + if not self.validate_master_theme_uniqueness(): + all_valid = False + + if not self.validate_charts(): + all_valid = False + + if not self.validate_slides(): + all_valid = False + + return all_valid + + def _package_map(self) -> dict: + wanted = [] + wanted += list(self.unpacked_dir.glob("[[]Content_Types[]].xml")) + wanted += list(self.unpacked_dir.glob("ppt/presentation.xml")) + wanted += list(self.unpacked_dir.glob("ppt/theme/*.xml")) + wanted += list(self.unpacked_dir.glob("ppt/theme/_rels/*.rels")) + wanted += list(self.unpacked_dir.glob("ppt/charts/chart*.xml")) + for group in ("slideMasters", "notesMasters", "handoutMasters"): + wanted += list(self.unpacked_dir.glob(f"ppt/{group}/*.xml")) + wanted += list(self.unpacked_dir.glob(f"ppt/{group}/_rels/*.rels")) + return { + p.relative_to(self.unpacked_dir).as_posix(): p.read_bytes() + for p in wanted + if p.is_file() + } + + def validate_master_theme_uniqueness(self): + from helpers.pptx_theme import _NOTES_MASTERS, live_shared_master_themes + + shared = live_shared_master_themes(self._package_map()) + if shared: + print(f"FAILED - Found {len(shared)} master(s) sharing a theme part:") + for message in shared: + print(f" {message}") + if any(m.startswith(_NOTES_MASTERS) for m in shared): + print(" Fix: in ppt/presentation.xml, move back to " + "directly after . PowerPoint reads that happily.") + else: + print(" Fix: give each master its own theme part.") + return False + + if self.verbose: + print("PASSED - No master shares a theme part in a way PowerPoint refuses") + return True + + def validate_charts(self): + from helpers.pptx_chart import find_chart_problems + + problems = find_chart_problems(self._package_map()) + if problems: + print(f"FAILED - Found {len(problems)} chart problem(s) PowerPoint rejects:") + for message in problems: + print(f" {message}") + return False + + if self.verbose: + print("PASSED - Charts satisfy the constraints PowerPoint enforces") + return True + + def _original_slide_defects(self, schema) -> set[str]: + import tempfile + import zipfile + + from helpers.pptx_slide import SLIDE_PART_RE, fatal_slide_errors + + if self.original_file is None: + return set() + + found: set[str] = set() + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + try: + with zipfile.ZipFile(self.original_file, "r") as zf: + safe_extract(zf, temp_path) + except (zipfile.BadZipFile, ValueError, OSError): + return set() + + for part in sorted(temp_path.rglob("*.xml")): + relative = part.relative_to(temp_path).as_posix() + if not SLIDE_PART_RE.fullmatch(relative): + continue + ok, errors = self._validate_single_file_xsd( + part.resolve(), temp_path.resolve(), schema_path=schema + ) + if ok is None or ok or not errors: + continue + found |= set(fatal_slide_errors(set(errors))) + return found + + def validate_slides(self): + from helpers.pptx_slide import ( + SLIDE_PART_RE, + fatal_slide_errors, + is_schema_verdict, + ) + + schema = self.schemas_dir / self.SCHEMA_MAPPINGS["ppt"] + inherited = self._original_slide_defects(schema) + problems: list[str] = [] + broken: list[str] = [] + + for xml_file in self.xml_files: + relative = xml_file.relative_to(self.unpacked_dir).as_posix() + if not SLIDE_PART_RE.fullmatch(relative): + continue + ok, errors = self._validate_single_file_xsd( + xml_file.resolve(), self.unpacked_dir.resolve(), schema_path=schema + ) + if ok is None or not errors: + continue + + unreadable = [f"{relative}: {e}" for e in errors if not is_schema_verdict(e)] + if unreadable: + broken.extend(unreadable) + continue + if ok: + continue + + for message in fatal_slide_errors(set(errors)): + if message in inherited: + continue + problems.append(f"{relative}: {message}") + + if broken: + print(f"FAILED - Could not check {len(broken)} slide part(s):") + for message in sorted(broken): + print(f" {message[:240]}") + + if problems: + print(f"FAILED - Found {len(problems)} slide problem(s) PowerPoint rejects:") + for message in sorted(problems): + print(f" {message[:240]}") + + if broken or problems: + return False + + if self.verbose: + print("PASSED - Slide XML has none of the defects PowerPoint refuses") + return True + + def _get_schema_path(self, xml_file): + if xml_file.parent.name == "charts" and xml_file.name.startswith("chart"): + return None + return super()._get_schema_path(xml_file) + + def _preprocess_for_schema(self, xml_doc, relative_path): + if relative_path.as_posix() != "ppt/presentation.xml": + return xml_doc + + root = xml_doc.getroot() + ns = f"{{{self.PRESENTATIONML_NAMESPACE}}}" + notes = root.find(f"{ns}notesMasterIdLst") + slides = root.find(f"{ns}sldIdLst") + if notes is None or slides is None: + return xml_doc + + children = list(root) + if children.index(notes) < children.index(slides): + return xml_doc + + root.remove(notes) + root.insert(list(root).index(slides), notes) + return xml_doc + + def validate_uuid_ids(self): + import lxml.etree + + errors = [] + uuid_pattern = re.compile( + r"^[\{\(]?[0-9A-Fa-f]{8}-?[0-9A-Fa-f]{4}-?[0-9A-Fa-f]{4}-?[0-9A-Fa-f]{4}-?[0-9A-Fa-f]{12}[\}\)]?$" + ) + + for xml_file in self.xml_files: + try: + root = lxml.etree.parse(str(xml_file)).getroot() + + for elem in root.iter(): + for attr, value in elem.attrib.items(): + attr_name = attr.split("}")[-1].lower() + if attr_name == "id" or attr_name.endswith("id"): + if self._looks_like_uuid(value): + if not uuid_pattern.match(value): + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: ID '{value}' appears to be a UUID but contains invalid hex characters" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} UUID ID validation errors:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All UUID-like IDs contain valid hex values") + return True + + def _looks_like_uuid(self, value): + clean_value = value.strip("{}()").replace("-", "") + return len(clean_value) == 32 and all(c.isalnum() for c in clean_value) + + def validate_slide_layout_ids(self): + import lxml.etree + + errors = [] + + slide_masters = list(self.unpacked_dir.glob("ppt/slideMasters/*.xml")) + + if not slide_masters: + if self.verbose: + print("PASSED - No slide masters found") + return True + + for slide_master in slide_masters: + try: + root = lxml.etree.parse(str(slide_master)).getroot() + + rels_file = slide_master.parent / "_rels" / f"{slide_master.name}.rels" + + if not rels_file.exists(): + errors.append( + f" {slide_master.relative_to(self.unpacked_dir)}: " + f"Missing relationships file: {rels_file.relative_to(self.unpacked_dir)}" + ) + continue + + rels_root = lxml.etree.parse(str(rels_file)).getroot() + + valid_layout_rids = set() + for rel in rels_root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ): + rel_type = rel.get("Type", "") + if "slideLayout" in rel_type: + valid_layout_rids.add(rel.get("Id")) + + for sld_layout_id in root.findall( + f".//{{{self.PRESENTATIONML_NAMESPACE}}}sldLayoutId" + ): + r_id = sld_layout_id.get( + f"{{{self.OFFICE_RELATIONSHIPS_NAMESPACE}}}id" + ) + layout_id = sld_layout_id.get("id") + + if r_id and r_id not in valid_layout_rids: + errors.append( + f" {slide_master.relative_to(self.unpacked_dir)}: " + f"Line {sld_layout_id.sourceline}: sldLayoutId with id='{layout_id}' " + f"references r:id='{r_id}' which is not found in slide layout relationships" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {slide_master.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} slide layout ID validation errors:") + for error in errors: + print(error) + print( + "Remove invalid references or add missing slide layouts to the relationships file." + ) + return False + else: + if self.verbose: + print("PASSED - All slide layout IDs reference valid slide layouts") + return True + + def validate_no_duplicate_slide_layouts(self): + import lxml.etree + + errors = [] + slide_rels_files = list(self.unpacked_dir.glob("ppt/slides/_rels/*.xml.rels")) + + for rels_file in slide_rels_files: + try: + root = lxml.etree.parse(str(rels_file)).getroot() + + layout_rels = [ + rel + for rel in root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ) + if "slideLayout" in rel.get("Type", "") + ] + + if len(layout_rels) > 1: + errors.append( + f" {rels_file.relative_to(self.unpacked_dir)}: has {len(layout_rels)} slideLayout references" + ) + + except Exception as e: + errors.append( + f" {rels_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print("FAILED - Found slides with duplicate slideLayout references:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All slides have exactly one slideLayout reference") + return True + + def validate_notes_slide_references(self): + import lxml.etree + + errors = [] + notes_slide_references = {} + + slide_rels_files = list(self.unpacked_dir.glob("ppt/slides/_rels/*.xml.rels")) + + if not slide_rels_files: + if self.verbose: + print("PASSED - No slide relationship files found") + return True + + for rels_file in slide_rels_files: + try: + root = lxml.etree.parse(str(rels_file)).getroot() + + for rel in root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ): + rel_type = rel.get("Type", "") + if "notesSlide" in rel_type: + part = opc_target( + rel.get("Target", ""), + rels_source_part(rels_file, self.unpacked_dir), + rel.get("TargetMode", ""), + ) + if part: + slide_name = rels_file.stem.replace( + ".xml", "" + ) + + notes_slide_references.setdefault(part, []).append( + (slide_name, rels_file) + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {rels_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + for target, references in notes_slide_references.items(): + if len(references) > 1: + slide_names = [ref[0] for ref in references] + errors.append( + f" Notes slide '{target}' is referenced by multiple slides: {', '.join(slide_names)}" + ) + for slide_name, rels_file in references: + errors.append(f" - {rels_file.relative_to(self.unpacked_dir)}") + + if errors: + print( + f"FAILED - Found {len([e for e in errors if not e.startswith(' ')])} notes slide reference validation errors:" + ) + for error in errors: + print(error) + print("Each slide may optionally have its own slide file.") + return False + else: + if self.verbose: + print("PASSED - All notes slide references are unique") + return True + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-docx/scripts/office/validators/redlining.py b/.github/skills/anthropic-docx/scripts/office/validators/redlining.py new file mode 100644 index 00000000..4185c51f --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/office/validators/redlining.py @@ -0,0 +1,299 @@ +""" +Validator for tracked changes in Word documents. + +Detects untracked edits in word/document.xml: text that differs from the +original without a / wrapper recording it. The tracked changes +that are new relative to the original are undone, and the result is compared +against the original; whatever text still differs was edited without being +tracked. + +Only the document body is compared. Headers, footers, footnotes and endnotes +are separate parts and are not checked. +""" + +import subprocess +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.ElementTree as ET +from defusedxml.common import DefusedXmlException + +from helpers import rendered_text, safe_extract + + +class RedliningValidator: + + def __init__(self, unpacked_dir, original_docx, verbose=False): + self.unpacked_dir = Path(unpacked_dir) + self.original_docx = Path(original_docx) + self.verbose = verbose + self.namespaces = { + "w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + } + + def repair(self) -> int: + return 0 + + def validate(self): + modified_file = self.unpacked_dir / "word" / "document.xml" + if not modified_file.exists(): + print(f"FAILED - Modified document.xml not found at {modified_file}") + return False + + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + + try: + with zipfile.ZipFile(self.original_docx, "r") as zip_ref: + safe_extract(zip_ref, temp_path) + except Exception as e: + print(f"FAILED - Error unpacking original docx: {e}") + return False + + original_file = temp_path / "word" / "document.xml" + if not original_file.exists(): + print( + f"FAILED - Original document.xml not found in {self.original_docx}" + ) + return False + + try: + modified_tree = ET.parse(modified_file) + modified_root = modified_tree.getroot() + original_tree = ET.parse(original_file) + original_root = original_tree.getroot() + except (ET.ParseError, DefusedXmlException) as e: + print(f"FAILED - Error parsing XML files: {e}") + return False + + new_changes = self._new_tracked_changes(original_root, modified_root) + self._remove_tracked_changes(modified_root, new_changes) + + modified_text = self._extract_text_content(modified_root) + original_text = self._extract_text_content(original_root) + + if modified_text != original_text: + error_message = self._generate_detailed_diff( + original_text, modified_text + ) + print(error_message) + return False + + if self.verbose: + print( + f"PASSED - All {len(new_changes)} change(s) against the original " + "are properly tracked" + ) + return True + + def _tracked_change_elements(self, root): + ins_tag = f"{{{self.namespaces['w']}}}ins" + del_tag = f"{{{self.namespaces['w']}}}del" + return [elem for elem in root.iter() if elem.tag in (ins_tag, del_tag)] + + def _rendered_text(self, elem): + preserve = elem.get("{http://www.w3.org/XML/1998/namespace}space") == "preserve" + return rendered_text(elem.text or "", preserve) + + def _text_elements(self, elem): + w = self.namespaces["w"] + return [ + node + for node in elem.iter() + if node.tag in (f"{{{w}}}t", f"{{{w}}}delText") + ] + + def _tracked_change_key(self, elem): + w = self.namespaces["w"] + text = "".join(self._rendered_text(node) for node in self._text_elements(elem)) + return (elem.tag, elem.get(f"{{{w}}}author"), elem.get(f"{{{w}}}date"), text) + + def _new_tracked_changes(self, original_root, modified_root): + original = self._tracked_change_elements(original_root) + modified = self._tracked_change_elements(modified_root) + + pool = {} + for elem in original: + pool.setdefault(self._tracked_change_key(elem), []).append(elem) + + matched, leftover = set(), [] + for elem in modified: + bucket = pool.get(self._tracked_change_key(elem)) + if bucket: + matched.add(bucket.pop()) + else: + leftover.append(elem) + + def group(elem): + return self._tracked_change_key(elem)[:3] + + def text_of(elems): + return "".join(self._tracked_change_key(e)[3] for e in elems) + + unmatched_original = {} + for elem in original: + if elem not in matched: + unmatched_original.setdefault(group(elem), []).append(elem) + + by_group = {} + for elem in leftover: + by_group.setdefault(group(elem), []).append(elem) + + new = set() + for key, elems in by_group.items(): + rebuilt = text_of(elems) + if rebuilt and rebuilt == text_of(unmatched_original.get(key, [])): + continue + new.update(elems) + return new + + def _generate_detailed_diff(self, original_text, modified_text): + error_parts = [ + "FAILED - Document text doesn't match after removing the tracked changes", + "", + "Likely causes:", + " 1. Modified text inside another author's or tags", + " 2. Made edits without proper tracked changes", + " 3. Didn't nest inside when deleting another's insertion", + " 4. Rewrote another author's / and changed its text on", + " the way. A tracked change from the original is recognised by its", + " author, date and text; anything that doesn't reproduce one exactly", + " reads as new, and the text it carried is reported missing.", + "", + "For pre-redlined documents, use correct patterns:", + " - To reject another's INSERTION: Nest inside their ", + " - To reject PART of one: nest around only the runs you reject.", + " Their may be split around it, so long as the pieces keep", + " their author and date and still spell out the same text.", + " - To restore another's DELETION: Add new AFTER their ", + "", + ] + + git_diff = self._get_git_word_diff(original_text, modified_text) + if git_diff: + error_parts.extend(["Differences:", "============", git_diff]) + else: + error_parts.append("Unable to generate word diff (git not available)") + + return "\n".join(error_parts) + + def _get_git_word_diff(self, original_text, modified_text): + try: + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + + original_file = temp_path / "original.txt" + modified_file = temp_path / "modified.txt" + + original_file.write_text(original_text, encoding="utf-8") + modified_file.write_text(modified_text, encoding="utf-8") + + result = subprocess.run( + [ + "git", + "diff", + "--word-diff=plain", + "--word-diff-regex=.", + "-U0", + "--no-index", + str(original_file), + str(modified_file), + ], + capture_output=True, + text=True, + ) + + if result.stdout.strip(): + lines = result.stdout.split("\n") + content_lines = [] + in_content = False + for line in lines: + if line.startswith("@@"): + in_content = True + continue + if in_content and line.strip(): + content_lines.append(line) + + if content_lines: + return "\n".join(content_lines) + + result = subprocess.run( + [ + "git", + "diff", + "--word-diff=plain", + "-U0", + "--no-index", + str(original_file), + str(modified_file), + ], + capture_output=True, + text=True, + ) + + if result.stdout.strip(): + lines = result.stdout.split("\n") + content_lines = [] + in_content = False + for line in lines: + if line.startswith("@@"): + in_content = True + continue + if in_content and line.strip(): + content_lines.append(line) + return "\n".join(content_lines) + + except (subprocess.CalledProcessError, FileNotFoundError, Exception): + pass + + return None + + def _remove_tracked_changes(self, root, targets): + ins_tag = f"{{{self.namespaces['w']}}}ins" + del_tag = f"{{{self.namespaces['w']}}}del" + + for parent in root.iter(): + to_remove = [] + for child in parent: + if child.tag == ins_tag and child in targets: + to_remove.append(child) + for elem in to_remove: + parent.remove(elem) + + deltext_tag = f"{{{self.namespaces['w']}}}delText" + t_tag = f"{{{self.namespaces['w']}}}t" + + for parent in root.iter(): + to_process = [] + for child in parent: + if child.tag == del_tag and child in targets: + to_process.append((child, list(parent).index(child))) + + for del_elem, del_index in reversed(to_process): + for elem in del_elem.iter(): + if elem.tag == deltext_tag: + elem.tag = t_tag + + for child in reversed(list(del_elem)): + parent.insert(del_index, child) + parent.remove(del_elem) + + def _extract_text_content(self, root): + p_tag = f"{{{self.namespaces['w']}}}p" + t_tag = f"{{{self.namespaces['w']}}}t" + + paragraphs = [] + for p_elem in root.findall(f".//{p_tag}"): + text_parts = [] + for t_elem in p_elem.findall(f".//{t_tag}"): + text_parts.append(self._rendered_text(t_elem)) + paragraph_text = "".join(text_parts) + if paragraph_text: + paragraphs.append(paragraph_text) + + return "\n".join(paragraphs) + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-docx/scripts/templates/comments.xml b/.github/skills/anthropic-docx/scripts/templates/comments.xml new file mode 100644 index 00000000..cd01a7d7 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/templates/comments.xml @@ -0,0 +1,3 @@ + + + diff --git a/.github/skills/anthropic-docx/scripts/templates/commentsExtended.xml b/.github/skills/anthropic-docx/scripts/templates/commentsExtended.xml new file mode 100644 index 00000000..411003cc --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/templates/commentsExtended.xml @@ -0,0 +1,3 @@ + + + diff --git a/.github/skills/anthropic-docx/scripts/templates/commentsExtensible.xml b/.github/skills/anthropic-docx/scripts/templates/commentsExtensible.xml new file mode 100644 index 00000000..f5572d71 --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/templates/commentsExtensible.xml @@ -0,0 +1,3 @@ + + + diff --git a/.github/skills/anthropic-docx/scripts/templates/commentsIds.xml b/.github/skills/anthropic-docx/scripts/templates/commentsIds.xml new file mode 100644 index 00000000..32f1629f --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/templates/commentsIds.xml @@ -0,0 +1,3 @@ + + + diff --git a/.github/skills/anthropic-docx/scripts/templates/people.xml b/.github/skills/anthropic-docx/scripts/templates/people.xml new file mode 100644 index 00000000..3803d2de --- /dev/null +++ b/.github/skills/anthropic-docx/scripts/templates/people.xml @@ -0,0 +1,3 @@ + + + diff --git a/.github/skills/anthropic-pdf/LICENSE.txt b/.github/skills/anthropic-pdf/LICENSE.txt new file mode 100644 index 00000000..c55ab422 --- /dev/null +++ b/.github/skills/anthropic-pdf/LICENSE.txt @@ -0,0 +1,30 @@ +© 2025 Anthropic, PBC. All rights reserved. + +LICENSE: Use of these materials (including all code, prompts, assets, files, +and other components of this Skill) is governed by your agreement with +Anthropic regarding use of Anthropic's services. If no separate agreement +exists, use is governed by Anthropic's Consumer Terms of Service or +Commercial Terms of Service, as applicable: +https://www.anthropic.com/legal/consumer-terms +https://www.anthropic.com/legal/commercial-terms +Your applicable agreement is referred to as the "Agreement." "Services" are +as defined in the Agreement. + +ADDITIONAL RESTRICTIONS: Notwithstanding anything in the Agreement to the +contrary, users may not: + +- Extract these materials from the Services or retain copies of these + materials outside the Services +- Reproduce or copy these materials, except for temporary copies created + automatically during authorized use of the Services +- Create derivative works based on these materials +- Distribute, sublicense, or transfer these materials to any third party +- Make, offer to sell, sell, or import any inventions embodied in these + materials +- Reverse engineer, decompile, or disassemble these materials + +The receipt, viewing, or possession of these materials does not convey or +imply any license or right beyond those expressly granted above. + +Anthropic retains all right, title, and interest in these materials, +including all copyrights, patents, and other intellectual property rights. diff --git a/.github/skills/anthropic-pdf/SKILL.md b/.github/skills/anthropic-pdf/SKILL.md new file mode 100644 index 00000000..5481d699 --- /dev/null +++ b/.github/skills/anthropic-pdf/SKILL.md @@ -0,0 +1,274 @@ +--- +name: anthropic-pdf +description: "Use when tasks involve reading, creating, or reviewing PDF files where rendering and layout matter; prefer visual checks by rendering pages (Poppler) and use Python tools such as `reportlab`, `pdfplumber`, and `pypdf` for generation and extraction." +--- + +# PDF Processing Guide + +## Overview + +This guide covers essential PDF processing operations using Python libraries and command-line tools. For advanced features, JavaScript libraries, and detailed examples, see REFERENCE.md. If you need to fill out a PDF form, read FORMS.md and follow its instructions. + +## Quick Start + +```python +from pypdf import PdfReader, PdfWriter + +reader = PdfReader("document.pdf") +print(f"Pages: {len(reader.pages)}") + +text = "" +for page in reader.pages: + text += page.extract_text() +``` + +## Python Libraries + +### pypdf - Basic Operations + +#### Merge PDFs +```python +from pypdf import PdfWriter, PdfReader + +writer = PdfWriter() +for pdf_file in ["doc1.pdf", "doc2.pdf", "doc3.pdf"]: + reader = PdfReader(pdf_file) + for page in reader.pages: + writer.add_page(page) + +with open("merged.pdf", "wb") as output: + writer.write(output) +``` + +#### Split PDF +```python +reader = PdfReader("input.pdf") +for i, page in enumerate(reader.pages): + writer = PdfWriter() + writer.add_page(page) + with open(f"page_{i+1}.pdf", "wb") as output: + writer.write(output) +``` + +#### Extract Metadata +```python +reader = PdfReader("document.pdf") +meta = reader.metadata +print(f"Title: {meta.title}") +print(f"Author: {meta.author}") +print(f"Subject: {meta.subject}") +print(f"Creator: {meta.creator}") +``` + +#### Rotate Pages +```python +from pypdf import PdfReader, PdfWriter + +reader = PdfReader("input.pdf") +writer = PdfWriter() + +page = reader.pages[0] +page.rotate(90) +writer.add_page(page) + +with open("rotated.pdf", "wb") as output: + writer.write(output) +``` + +### pdfplumber - Text and Table Extraction + +#### Extract Text with Layout +```python +import pdfplumber + +with pdfplumber.open("document.pdf") as pdf: + for page in pdf.pages: + text = page.extract_text() + print(text) +``` + +#### Extract Tables +```python +with pdfplumber.open("document.pdf") as pdf: + for i, page in enumerate(pdf.pages): + tables = page.extract_tables() + for j, table in enumerate(tables): + print(f"Table {j+1} on page {i+1}:") + for row in table: + print(row) +``` + +#### Advanced Table Extraction +```python +import pandas as pd + +with pdfplumber.open("document.pdf") as pdf: + all_tables = [] + for page in pdf.pages: + tables = page.extract_tables() + for table in tables: + if table: + df = pd.DataFrame(table[1:], columns=table[0]) + all_tables.append(df) + +if all_tables: + combined_df = pd.concat(all_tables, ignore_index=True) + combined_df.to_excel("extracted_tables.xlsx", index=False) +``` + +### reportlab - Create PDFs + +#### Basic PDF Creation +```python +from reportlab.lib.pagesizes import letter +from reportlab.pdfgen import canvas + +c = canvas.Canvas("hello.pdf", pagesize=letter) +width, height = letter + +c.drawString(100, height - 100, "Hello World!") +c.drawString(100, height - 120, "This is a PDF created with reportlab") +c.line(100, height - 140, 400, height - 140) + +c.save() +``` + +#### Create PDF with Multiple Pages +```python +from reportlab.lib.pagesizes import letter +from reportlab.platypus import SimpleDocTemplate, Paragraph, Spacer, PageBreak +from reportlab.lib.styles import getSampleStyleSheet + +doc = SimpleDocTemplate("report.pdf", pagesize=letter) +styles = getSampleStyleSheet() +story = [] + +title = Paragraph("Report Title", styles['Title']) +story.append(title) +story.append(Spacer(1, 12)) + +body = Paragraph("This is the body of the report. " * 20, styles['Normal']) +story.append(body) +story.append(PageBreak()) + +story.append(Paragraph("Page 2", styles['Heading1'])) +story.append(Paragraph("Content for page 2", styles['Normal'])) + +doc.build(story) +``` + +#### Subscripts and Superscripts + +**IMPORTANT**: Never use Unicode subscript/superscript characters in ReportLab PDFs. The built-in fonts do not include these glyphs, causing them to render as solid black boxes. + +Instead, use ReportLab's XML markup tags in Paragraph objects: +```python +from reportlab.platypus import Paragraph +from reportlab.lib.styles import getSampleStyleSheet + +styles = getSampleStyleSheet() + +chemical = Paragraph("H2O", styles['Normal']) +squared = Paragraph("x2 + y2", styles['Normal']) +``` + +## Command-Line Tools + +### pdftotext (poppler-utils) +```bash +pdftotext input.pdf output.txt +pdftotext -layout input.pdf output.txt +pdftotext -f 1 -l 5 input.pdf output.txt +``` + +### qpdf +```bash +qpdf --empty --pages file1.pdf file2.pdf -- merged.pdf +qpdf input.pdf --pages . 1-5 -- pages1-5.pdf +qpdf input.pdf --pages . 6-10 -- pages6-10.pdf +qpdf input.pdf output.pdf --rotate=+90:1 +qpdf --password=mypassword --decrypt encrypted.pdf decrypted.pdf +``` + +### pdftk (if available) +```bash +pdftk file1.pdf file2.pdf cat output merged.pdf +pdftk input.pdf burst +pdftk input.pdf rotate 1east output rotated.pdf +``` + +## Common Tasks + +### Extract Text from Scanned PDFs +```python +import pytesseract +from pdf2image import convert_from_path + +images = convert_from_path('scanned.pdf') + +text = "" +for i, image in enumerate(images): + text += f"Page {i+1}:\n" + text += pytesseract.image_to_string(image) + text += "\n\n" + +print(text) +``` + +### Add Watermark +```python +from pypdf import PdfReader, PdfWriter + +watermark = PdfReader("watermark.pdf").pages[0] + +reader = PdfReader("document.pdf") +writer = PdfWriter() + +for page in reader.pages: + page.merge_page(watermark) + writer.add_page(page) + +with open("watermarked.pdf", "wb") as output: + writer.write(output) +``` + +### Extract Images +```bash +pdfimages -j input.pdf output_prefix +``` + +### Password Protection +```python +from pypdf import PdfReader, PdfWriter + +reader = PdfReader("input.pdf") +writer = PdfWriter() + +for page in reader.pages: + writer.add_page(page) + +writer.encrypt("userpassword", "ownerpassword") + +with open("encrypted.pdf", "wb") as output: + writer.write(output) +``` + +## Quick Reference + +| Task | Best Tool | Command/Code | +|------|-----------|--------------| +| Merge PDFs | pypdf | `writer.add_page(page)` | +| Split PDFs | pypdf | One page per file | +| Extract text | pdfplumber | `page.extract_text()` | +| Extract tables | pdfplumber | `page.extract_tables()` | +| Create PDFs | reportlab | Canvas or Platypus | +| Command line merge | qpdf | `qpdf --empty --pages ...` | +| OCR scanned PDFs | pytesseract | Convert to image first | +| Fill PDF forms | pdf-lib or pypdf (see FORMS.md) | See FORMS.md | + +## Next Steps + +- For advanced pypdfium2 usage, see REFERENCE.md +- For JavaScript libraries (pdf-lib), see REFERENCE.md +- If you need to fill out a PDF form, follow the instructions in FORMS.md +- For troubleshooting guides, see REFERENCE.md diff --git a/.github/skills/anthropic-pdf/forms.md b/.github/skills/anthropic-pdf/forms.md new file mode 100644 index 00000000..5f634ab9 --- /dev/null +++ b/.github/skills/anthropic-pdf/forms.md @@ -0,0 +1,288 @@ +**CRITICAL: You MUST complete these steps in order. Do not skip ahead to writing code.** + +If you need to fill out a PDF form, first check to see if the PDF has fillable form fields. Run this script from this file's directory: + `python scripts/check_fillable_fields `, and depending on the result go to either the "Fillable fields" or "Non-fillable fields" and follow those instructions. + +# Fillable fields +If the PDF has fillable form fields: +- Run this script from this file's directory: `python scripts/extract_form_field_info.py `. It will create a JSON file with a list of fields in this format: +``` +[ + { + "field_id": (unique ID for the field), + "page": (page number, 1-based), + "rect": ([left, bottom, right, top] bounding box in PDF coordinates, y=0 is the bottom of the page), + "type": ("text", "checkbox", "radio_group", or "choice"), + }, + { + "field_id": (unique ID for the field), + "page": (page number, 1-based), + "type": "checkbox", + "checked_value": (Set the field to this value to check the checkbox), + "unchecked_value": (Set the field to this value to uncheck the checkbox), + }, + { + "field_id": (unique ID for the field), + "page": (page number, 1-based), + "type": "radio_group", + "radio_options": [ + { + "value": (set the field to this value to select this radio option), + "rect": (bounding box for the radio button for this option) + }, + ] + }, + { + "field_id": (unique ID for the field), + "page": (page number, 1-based), + "type": "choice", + "choice_options": [ + { + "value": (set the field to this value to select this option), + "text": (display text of the option) + }, + ], + } +] +``` +- Convert the PDF to PNGs (one image for each page) with this script (run from this file's directory): +`python scripts/convert_pdf_to_images.py ` +Then analyze the images to determine the purpose of each form field (make sure to convert the bounding box PDF coordinates to image coordinates). +- Create a `field_values.json` file in this format with the values to be entered for each field: +``` +[ + { + "field_id": "last_name", + "description": "The user's last name", + "page": 1, + "value": "Simpson" + }, + { + "field_id": "Checkbox12", + "description": "Checkbox to be checked if the user is 18 or over", + "page": 1, + "value": "/On" + }, +] +``` +- Run the `fill_fillable_fields.py` script from this file's directory to create a filled-in PDF: +`python scripts/fill_fillable_fields.py ` +This script will verify that the field IDs and values you provide are valid; if it prints error messages, correct the appropriate fields and try again. + +# Non-fillable fields +If the PDF doesn't have fillable form fields, you'll add text annotations. First try to extract coordinates from the PDF structure (more accurate), then fall back to visual estimation if needed. + +## Step 1: Try Structure Extraction First + +Run this script to extract text labels, lines, and checkboxes with their exact PDF coordinates: +`python scripts/extract_form_structure.py form_structure.json` + +This creates a JSON file containing: +- **labels**: Every text element with exact coordinates (x0, top, x1, bottom in PDF points) +- **lines**: Horizontal lines that define row boundaries +- **checkboxes**: Small square rectangles that are checkboxes (with center coordinates) +- **row_boundaries**: Row top/bottom positions calculated from horizontal lines + +**Check the results**: If `form_structure.json` has meaningful labels (text elements that correspond to form fields), use **Approach A: Structure-Based Coordinates**. If the PDF is scanned/image-based and has few or no labels, use **Approach B: Visual Estimation**. + +--- + +## Approach A: Structure-Based Coordinates (Preferred) + +Use this when `extract_form_structure.py` found text labels in the PDF. + +### A.1: Analyze the Structure + +Read form_structure.json and identify: + +1. **Label groups**: Adjacent text elements that form a single label (e.g., "Last" + "Name") +2. **Row structure**: Labels with similar `top` values are in the same row +3. **Field columns**: Entry areas start after label ends (x0 = label.x1 + gap) +4. **Checkboxes**: Use the checkbox coordinates directly from the structure + +**Coordinate system**: PDF coordinates where y=0 is at TOP of page, y increases downward. + +### A.2: Check for Missing Elements + +The structure extraction may not detect all form elements. Common cases: +- **Circular checkboxes**: Only square rectangles are detected as checkboxes +- **Complex graphics**: Decorative elements or non-standard form controls +- **Faded or light-colored elements**: May not be extracted + +If you see form fields in the PDF images that aren't in form_structure.json, you'll need to use **visual analysis** for those specific fields (see "Hybrid Approach" below). + +### A.3: Create fields.json with PDF Coordinates + +For each field, calculate entry coordinates from the extracted structure: + +**Text fields:** +- entry x0 = label x1 + 5 (small gap after label) +- entry x1 = next label's x0, or row boundary +- entry top = same as label top +- entry bottom = row boundary line below, or label bottom + row_height + +**Checkboxes:** +- Use the checkbox rectangle coordinates directly from form_structure.json +- entry_bounding_box = [checkbox.x0, checkbox.top, checkbox.x1, checkbox.bottom] + +Create fields.json using `pdf_width` and `pdf_height` (signals PDF coordinates): +```json +{ + "pages": [ + {"page_number": 1, "pdf_width": 612, "pdf_height": 792} + ], + "form_fields": [ + { + "page_number": 1, + "description": "Last name entry field", + "field_label": "Last Name", + "label_bounding_box": [43, 63, 87, 73], + "entry_bounding_box": [92, 63, 260, 79], + "entry_text": {"text": "Smith", "font_size": 10} + }, + { + "page_number": 1, + "description": "US Citizen Yes checkbox", + "field_label": "Yes", + "label_bounding_box": [260, 200, 280, 210], + "entry_bounding_box": [285, 197, 292, 205], + "entry_text": {"text": "X"} + } + ] +} +``` + +**Important**: Use `pdf_width`/`pdf_height` and coordinates directly from form_structure.json. + +### A.4: Validate Bounding Boxes + +Before filling, check your bounding boxes for errors: +`python scripts/check_bounding_boxes.py fields.json` + +This checks for intersecting bounding boxes and entry boxes that are too small for the font size. Fix any reported errors before filling. + +--- + +## Approach B: Visual Estimation (Fallback) + +Use this when the PDF is scanned/image-based and structure extraction found no usable text labels (e.g., all text shows as "(cid:X)" patterns). + +### B.1: Convert PDF to Images + +`python scripts/convert_pdf_to_images.py ` + +### B.2: Initial Field Identification + +Examine each page image to identify form sections and get **rough estimates** of field locations: +- Form field labels and their approximate positions +- Entry areas (lines, boxes, or blank spaces for text input) +- Checkboxes and their approximate locations + +For each field, note approximate pixel coordinates (they don't need to be precise yet). + +### B.3: Zoom Refinement (CRITICAL for accuracy) + +For each field, crop a region around the estimated position to refine coordinates precisely. + +**Create a zoomed crop using ImageMagick:** +```bash +magick -crop x++ +repage +``` + +Where: +- `, ` = top-left corner of crop region (use your rough estimate minus padding) +- `, ` = size of crop region (field area plus ~50px padding on each side) + +**Example:** To refine a "Name" field estimated around (100, 150): +```bash +magick images_dir/page_1.png -crop 300x80+50+120 +repage crops/name_field.png +``` + +(Note: if the `magick` command isn't available, try `convert` with the same arguments). + +**Examine the cropped image** to determine precise coordinates: +1. Identify the exact pixel where the entry area begins (after the label) +2. Identify where the entry area ends (before next field or edge) +3. Identify the top and bottom of the entry line/box + +**Convert crop coordinates back to full image coordinates:** +- full_x = crop_x + crop_offset_x +- full_y = crop_y + crop_offset_y + +Example: If the crop started at (50, 120) and the entry box starts at (52, 18) within the crop: +- entry_x0 = 52 + 50 = 102 +- entry_top = 18 + 120 = 138 + +**Repeat for each field**, grouping nearby fields into single crops when possible. + +### B.4: Create fields.json with Refined Coordinates + +Create fields.json using `image_width` and `image_height` (signals image coordinates): +```json +{ + "pages": [ + {"page_number": 1, "image_width": 1700, "image_height": 2200} + ], + "form_fields": [ + { + "page_number": 1, + "description": "Last name entry field", + "field_label": "Last Name", + "label_bounding_box": [120, 175, 242, 198], + "entry_bounding_box": [255, 175, 720, 218], + "entry_text": {"text": "Smith", "font_size": 10} + } + ] +} +``` + +**Important**: Use `image_width`/`image_height` and the refined pixel coordinates from the zoom analysis. + +### B.5: Validate Bounding Boxes + +Before filling, check your bounding boxes for errors: +`python scripts/check_bounding_boxes.py fields.json` + +This checks for intersecting bounding boxes and entry boxes that are too small for the font size. Fix any reported errors before filling. + +--- + +## Hybrid Approach: Structure + Visual + +Use this when structure extraction works for most fields but misses some elements (e.g., circular checkboxes, unusual form controls). + +1. **Use Approach A** for fields that were detected in form_structure.json +2. **Convert PDF to images** for visual analysis of missing fields +3. **Use zoom refinement** (from Approach B) for the missing fields +4. **Combine coordinates**: For fields from structure extraction, use `pdf_width`/`pdf_height`. For visually-estimated fields, you must convert image coordinates to PDF coordinates: + - pdf_x = image_x * (pdf_width / image_width) + - pdf_y = image_y * (pdf_height / image_height) +5. **Use a single coordinate system** in fields.json - convert all to PDF coordinates with `pdf_width`/`pdf_height` + +--- + +## Step 2: Validate Before Filling + +**Always validate bounding boxes before filling:** +`python scripts/check_bounding_boxes.py fields.json` + +This checks for: +- Intersecting bounding boxes (which would cause overlapping text) +- Entry boxes that are too small for the specified font size + +Fix any reported errors in fields.json before proceeding. + +## Step 3: Fill the Form + +The fill script auto-detects the coordinate system and handles conversion: +`python scripts/fill_pdf_form_with_annotations.py fields.json ` + +## Step 4: Verify Output + +Convert the filled PDF to images and verify text placement: +`python scripts/convert_pdf_to_images.py ` + +If text is mispositioned: +- **Approach A**: Check that you're using PDF coordinates from form_structure.json with `pdf_width`/`pdf_height` +- **Approach B**: Check that image dimensions match and coordinates are accurate pixels +- **Hybrid**: Ensure coordinate conversions are correct for visually-estimated fields diff --git a/.github/skills/anthropic-pdf/reference.md b/.github/skills/anthropic-pdf/reference.md new file mode 100644 index 00000000..3012f4d1 --- /dev/null +++ b/.github/skills/anthropic-pdf/reference.md @@ -0,0 +1,535 @@ +# PDF Processing Advanced Reference + +This document contains advanced PDF processing features, detailed examples, and additional libraries not covered in the main skill instructions. + +## pypdfium2 Library (Apache/BSD License) + +### Overview +pypdfium2 is a Python binding for PDFium (Chromium's PDF library). It's excellent for fast PDF rendering, image generation, and serves as a PyMuPDF replacement. + +### Render PDF to Images +```python +import pypdfium2 as pdfium +from PIL import Image + +pdf = pdfium.PdfDocument("document.pdf") + +page = pdf[0] +bitmap = page.render( + scale=2.0, + rotation=0 +) + +img = bitmap.to_pil() +img.save("page_1.png", "PNG") + +for i, page in enumerate(pdf): + bitmap = page.render(scale=1.5) + img = bitmap.to_pil() + img.save(f"page_{i+1}.jpg", "JPEG", quality=90) +``` + +### Extract Text with pypdfium2 +```python +import pypdfium2 as pdfium + +pdf = pdfium.PdfDocument("document.pdf") +for i, page in enumerate(pdf): + text = page.get_text() + print(f"Page {i+1} text length: {len(text)} chars") +``` + +## JavaScript Libraries + +### pdf-lib (MIT License) + +pdf-lib is a powerful JavaScript library for creating and modifying PDF documents in any JavaScript environment. + +#### Load and Manipulate Existing PDF +```javascript +import { PDFDocument } from 'pdf-lib'; +import fs from 'fs'; + +async function manipulatePDF() { + const existingPdfBytes = fs.readFileSync('input.pdf'); + const pdfDoc = await PDFDocument.load(existingPdfBytes); + + const pageCount = pdfDoc.getPageCount(); + console.log(`Document has ${pageCount} pages`); + + const newPage = pdfDoc.addPage([600, 400]); + newPage.drawText('Added by pdf-lib', { + x: 100, + y: 300, + size: 16 + }); + + const pdfBytes = await pdfDoc.save(); + fs.writeFileSync('modified.pdf', pdfBytes); +} +``` + +#### Create Complex PDFs from Scratch +```javascript +import { PDFDocument, rgb, StandardFonts } from 'pdf-lib'; +import fs from 'fs'; + +async function createPDF() { + const pdfDoc = await PDFDocument.create(); + + const helveticaFont = await pdfDoc.embedFont(StandardFonts.Helvetica); + const helveticaBold = await pdfDoc.embedFont(StandardFonts.HelveticaBold); + + const page = pdfDoc.addPage([595, 842]); + const { width, height } = page.getSize(); + + page.drawText('Invoice #12345', { + x: 50, + y: height - 50, + size: 18, + font: helveticaBold, + color: rgb(0.2, 0.2, 0.8) + }); + + page.drawRectangle({ + x: 40, + y: height - 100, + width: width - 80, + height: 30, + color: rgb(0.9, 0.9, 0.9) + }); + + const items = [ + ['Item', 'Qty', 'Price', 'Total'], + ['Widget', '2', '$50', '$100'], + ['Gadget', '1', '$75', '$75'] + ]; + + let yPos = height - 150; + items.forEach(row => { + let xPos = 50; + row.forEach(cell => { + page.drawText(cell, { + x: xPos, + y: yPos, + size: 12, + font: helveticaFont + }); + xPos += 120; + }); + yPos -= 25; + }); + + const pdfBytes = await pdfDoc.save(); + fs.writeFileSync('created.pdf', pdfBytes); +} +``` + +#### Advanced Merge and Split Operations +```javascript +import { PDFDocument } from 'pdf-lib'; +import fs from 'fs'; + +async function mergePDFs() { + const mergedPdf = await PDFDocument.create(); + + const pdf1Bytes = fs.readFileSync('doc1.pdf'); + const pdf2Bytes = fs.readFileSync('doc2.pdf'); + + const pdf1 = await PDFDocument.load(pdf1Bytes); + const pdf2 = await PDFDocument.load(pdf2Bytes); + + const pdf1Pages = await mergedPdf.copyPages(pdf1, pdf1.getPageIndices()); + pdf1Pages.forEach(page => mergedPdf.addPage(page)); + + const pdf2Pages = await mergedPdf.copyPages(pdf2, [0, 2, 4]); + pdf2Pages.forEach(page => mergedPdf.addPage(page)); + + const mergedPdfBytes = await mergedPdf.save(); + fs.writeFileSync('merged.pdf', mergedPdfBytes); +} +``` + +### pdfjs-dist (Apache License) + +PDF.js is Mozilla's JavaScript library for rendering PDFs in the browser. + +#### Basic PDF Loading and Rendering +```javascript +import * as pdfjsLib from 'pdfjs-dist'; + +pdfjsLib.GlobalWorkerOptions.workerSrc = './pdf.worker.js'; + +async function renderPDF() { + const loadingTask = pdfjsLib.getDocument('document.pdf'); + const pdf = await loadingTask.promise; + + console.log(`Loaded PDF with ${pdf.numPages} pages`); + + const page = await pdf.getPage(1); + const viewport = page.getViewport({ scale: 1.5 }); + + const canvas = document.createElement('canvas'); + const context = canvas.getContext('2d'); + canvas.height = viewport.height; + canvas.width = viewport.width; + + const renderContext = { + canvasContext: context, + viewport: viewport + }; + + await page.render(renderContext).promise; + document.body.appendChild(canvas); +} +``` + +#### Extract Text with Coordinates +```javascript +import * as pdfjsLib from 'pdfjs-dist'; + +async function extractText() { + const loadingTask = pdfjsLib.getDocument('document.pdf'); + const pdf = await loadingTask.promise; + + let fullText = ''; + + for (let i = 1; i <= pdf.numPages; i++) { + const page = await pdf.getPage(i); + const textContent = await page.getTextContent(); + + const pageText = textContent.items + .map(item => item.str) + .join(' '); + + fullText += `\n--- Page ${i} ---\n${pageText}`; + + const textWithCoords = textContent.items.map(item => ({ + text: item.str, + x: item.transform[4], + y: item.transform[5], + width: item.width, + height: item.height + })); + } + + console.log(fullText); + return fullText; +} +``` + +#### Extract Annotations and Forms +```javascript +import * as pdfjsLib from 'pdfjs-dist'; + +async function extractAnnotations() { + const loadingTask = pdfjsLib.getDocument('annotated.pdf'); + const pdf = await loadingTask.promise; + + for (let i = 1; i <= pdf.numPages; i++) { + const page = await pdf.getPage(i); + const annotations = await page.getAnnotations(); + + annotations.forEach(annotation => { + console.log(`Annotation type: ${annotation.subtype}`); + console.log(`Content: ${annotation.contents}`); + console.log(`Coordinates: ${JSON.stringify(annotation.rect)}`); + }); + } +} +``` + +## Advanced Command-Line Operations + +### poppler-utils Advanced Features + +#### Extract Text with Bounding Box Coordinates +```bash +pdftotext -bbox-layout document.pdf output.xml +``` + +#### Advanced Image Conversion +```bash +pdftoppm -png -r 300 document.pdf output_prefix +pdftoppm -png -r 600 -f 1 -l 3 document.pdf high_res_pages +pdftoppm -jpeg -jpegopt quality=85 -r 200 document.pdf jpeg_output +``` + +#### Extract Embedded Images +```bash +pdfimages -j -p document.pdf page_images +pdfimages -list document.pdf +pdfimages -all document.pdf images/img +``` + +### qpdf Advanced Features + +#### Complex Page Manipulation +```bash +qpdf --split-pages=3 input.pdf output_group_%02d.pdf +qpdf input.pdf --pages input.pdf 1,3-5,8,10-end -- extracted.pdf +qpdf --empty --pages doc1.pdf 1-3 doc2.pdf 5-7 doc3.pdf 2,4 -- combined.pdf +``` + +#### PDF Optimization and Repair +```bash +qpdf --linearize input.pdf optimized.pdf +qpdf --optimize-level=all input.pdf compressed.pdf +qpdf --check input.pdf +qpdf --fix-qdf damaged.pdf repaired.pdf +qpdf --show-all-pages input.pdf > structure.txt +``` + +#### Advanced Encryption +```bash +qpdf --encrypt user_pass owner_pass 256 --print=none --modify=none -- input.pdf encrypted.pdf +qpdf --show-encryption encrypted.pdf +qpdf --password=secret123 --decrypt encrypted.pdf decrypted.pdf +``` + +## Advanced Python Techniques + +### pdfplumber Advanced Features + +#### Extract Text with Precise Coordinates +```python +import pdfplumber + +with pdfplumber.open("document.pdf") as pdf: + page = pdf.pages[0] + + chars = page.chars + for char in chars[:10]: + print(f"Char: '{char['text']}' at x:{char['x0']:.1f} y:{char['y0']:.1f}") + + bbox_text = page.within_bbox((100, 100, 400, 200)).extract_text() +``` + +#### Advanced Table Extraction with Custom Settings +```python +import pdfplumber +import pandas as pd + +with pdfplumber.open("complex_table.pdf") as pdf: + page = pdf.pages[0] + + table_settings = { + "vertical_strategy": "lines", + "horizontal_strategy": "lines", + "snap_tolerance": 3, + "intersection_tolerance": 15 + } + tables = page.extract_tables(table_settings) + + img = page.to_image(resolution=150) + img.save("debug_layout.png") +``` + +### reportlab Advanced Features + +#### Create Professional Reports with Tables +```python +from reportlab.platypus import SimpleDocTemplate, Table, TableStyle, Paragraph +from reportlab.lib.styles import getSampleStyleSheet +from reportlab.lib import colors + +data = [ + ['Product', 'Q1', 'Q2', 'Q3', 'Q4'], + ['Widgets', '120', '135', '142', '158'], + ['Gadgets', '85', '92', '98', '105'] +] + +doc = SimpleDocTemplate("report.pdf") +elements = [] + +styles = getSampleStyleSheet() +title = Paragraph("Quarterly Sales Report", styles['Title']) +elements.append(title) + +table = Table(data) +table.setStyle(TableStyle([ + ('BACKGROUND', (0, 0), (-1, 0), colors.grey), + ('TEXTCOLOR', (0, 0), (-1, 0), colors.whitesmoke), + ('ALIGN', (0, 0), (-1, -1), 'CENTER'), + ('FONTNAME', (0, 0), (-1, 0), 'Helvetica-Bold'), + ('FONTSIZE', (0, 0), (-1, 0), 14), + ('BOTTOMPADDING', (0, 0), (-1, 0), 12), + ('BACKGROUND', (0, 1), (-1, -1), colors.beige), + ('GRID', (0, 0), (-1, -1), 1, colors.black) +])) +elements.append(table) + +doc.build(elements) +``` + +## Complex Workflows + +### Extract Figures/Images from PDF + +#### Method 1: Using pdfimages (fastest) +```bash +pdfimages -all document.pdf images/img +``` + +#### Method 2: Using pypdfium2 + Image Processing +```python +import pypdfium2 as pdfium +from PIL import Image +import numpy as np + +def extract_figures(pdf_path, output_dir): + pdf = pdfium.PdfDocument(pdf_path) + + for page_num, page in enumerate(pdf): + bitmap = page.render(scale=3.0) + img = bitmap.to_pil() + + img_array = np.array(img) + + mask = np.any(img_array != [255, 255, 255], axis=2) +``` + +### Batch PDF Processing with Error Handling +```python +import os +import glob +from pypdf import PdfReader, PdfWriter +import logging + +logging.basicConfig(level=logging.INFO) +logger = logging.getLogger(__name__) + +def batch_process_pdfs(input_dir, operation='merge'): + pdf_files = glob.glob(os.path.join(input_dir, "*.pdf")) + + if operation == 'merge': + writer = PdfWriter() + for pdf_file in pdf_files: + try: + reader = PdfReader(pdf_file) + for page in reader.pages: + writer.add_page(page) + logger.info(f"Processed: {pdf_file}") + except Exception as e: + logger.error(f"Failed to process {pdf_file}: {e}") + continue + + with open("batch_merged.pdf", "wb") as output: + writer.write(output) + + elif operation == 'extract_text': + for pdf_file in pdf_files: + try: + reader = PdfReader(pdf_file) + text = "" + for page in reader.pages: + text += page.extract_text() + + output_file = pdf_file.replace('.pdf', '.txt') + with open(output_file, 'w', encoding='utf-8') as f: + f.write(text) + logger.info(f"Extracted text from: {pdf_file}") + + except Exception as e: + logger.error(f"Failed to extract text from {pdf_file}: {e}") + continue +``` + +### Advanced PDF Cropping +```python +from pypdf import PdfWriter, PdfReader + +reader = PdfReader("input.pdf") +writer = PdfWriter() + +page = reader.pages[0] +page.mediabox.left = 50 +page.mediabox.bottom = 50 +page.mediabox.right = 550 +page.mediabox.top = 750 + +writer.add_page(page) +with open("cropped.pdf", "wb") as output: + writer.write(output) +``` + +## Performance Optimization Tips + +### 1. For Large PDFs +- Use streaming approaches instead of loading entire PDF in memory +- Use `qpdf --split-pages` for splitting large files +- Process pages individually with pypdfium2 + +### 2. For Text Extraction +- `pdftotext -bbox-layout` is fastest for plain text extraction +- Use pdfplumber for structured data and tables +- Avoid `pypdf.extract_text()` for very large documents + +### 3. For Image Extraction +- `pdfimages` is much faster than rendering pages +- Use low resolution for previews, high resolution for final output + +### 4. For Form Filling +- pdf-lib maintains form structure better than most alternatives +- Pre-validate form fields before processing + +### 5. Memory Management +```python +def process_large_pdf(pdf_path, chunk_size=10): + reader = PdfReader(pdf_path) + total_pages = len(reader.pages) + + for start_idx in range(0, total_pages, chunk_size): + end_idx = min(start_idx + chunk_size, total_pages) + writer = PdfWriter() + + for i in range(start_idx, end_idx): + writer.add_page(reader.pages[i]) + + with open(f"chunk_{start_idx//chunk_size}.pdf", "wb") as output: + writer.write(output) +``` + +## Troubleshooting Common Issues + +### Encrypted PDFs +```python +from pypdf import PdfReader + +try: + reader = PdfReader("encrypted.pdf") + if reader.is_encrypted: + reader.decrypt("password") +except Exception as e: + print(f"Failed to decrypt: {e}") +``` + +### Corrupted PDFs +```bash +qpdf --check corrupted.pdf +qpdf --replace-input corrupted.pdf +``` + +### Text Extraction Issues +```python +import pytesseract +from pdf2image import convert_from_path + +def extract_text_with_ocr(pdf_path): + images = convert_from_path(pdf_path) + text = "" + for i, image in enumerate(images): + text += pytesseract.image_to_string(image) + return text +``` + +## License Information + +- **pypdf**: BSD License +- **pdfplumber**: MIT License +- **pypdfium2**: Apache/BSD License +- **reportlab**: BSD License +- **poppler-utils**: GPL-2 License +- **qpdf**: Apache License +- **pdf-lib**: MIT License +- **pdfjs-dist**: Apache License diff --git a/.github/skills/anthropic-pdf/scripts/check_bounding_boxes.py b/.github/skills/anthropic-pdf/scripts/check_bounding_boxes.py new file mode 100644 index 00000000..2cc5e348 --- /dev/null +++ b/.github/skills/anthropic-pdf/scripts/check_bounding_boxes.py @@ -0,0 +1,65 @@ +from dataclasses import dataclass +import json +import sys + + + + +@dataclass +class RectAndField: + rect: list[float] + rect_type: str + field: dict + + +def get_bounding_box_messages(fields_json_stream) -> list[str]: + messages = [] + fields = json.load(fields_json_stream) + messages.append(f"Read {len(fields['form_fields'])} fields") + + def rects_intersect(r1, r2): + disjoint_horizontal = r1[0] >= r2[2] or r1[2] <= r2[0] + disjoint_vertical = r1[1] >= r2[3] or r1[3] <= r2[1] + return not (disjoint_horizontal or disjoint_vertical) + + rects_and_fields = [] + for f in fields["form_fields"]: + rects_and_fields.append(RectAndField(f["label_bounding_box"], "label", f)) + rects_and_fields.append(RectAndField(f["entry_bounding_box"], "entry", f)) + + has_error = False + for i, ri in enumerate(rects_and_fields): + for j in range(i + 1, len(rects_and_fields)): + rj = rects_and_fields[j] + if ri.field["page_number"] == rj.field["page_number"] and rects_intersect(ri.rect, rj.rect): + has_error = True + if ri.field is rj.field: + messages.append(f"FAILURE: intersection between label and entry bounding boxes for `{ri.field['description']}` ({ri.rect}, {rj.rect})") + else: + messages.append(f"FAILURE: intersection between {ri.rect_type} bounding box for `{ri.field['description']}` ({ri.rect}) and {rj.rect_type} bounding box for `{rj.field['description']}` ({rj.rect})") + if len(messages) >= 20: + messages.append("Aborting further checks; fix bounding boxes and try again") + return messages + if ri.rect_type == "entry": + if "entry_text" in ri.field: + font_size = ri.field["entry_text"].get("font_size", 14) + entry_height = ri.rect[3] - ri.rect[1] + if entry_height < font_size: + has_error = True + messages.append(f"FAILURE: entry bounding box height ({entry_height}) for `{ri.field['description']}` is too short for the text content (font size: {font_size}). Increase the box height or decrease the font size.") + if len(messages) >= 20: + messages.append("Aborting further checks; fix bounding boxes and try again") + return messages + + if not has_error: + messages.append("SUCCESS: All bounding boxes are valid") + return messages + +if __name__ == "__main__": + if len(sys.argv) != 2: + print("Usage: check_bounding_boxes.py [fields.json]") + sys.exit(1) + with open(sys.argv[1]) as f: + messages = get_bounding_box_messages(f) + for msg in messages: + print(msg) diff --git a/.github/skills/anthropic-pdf/scripts/check_fillable_fields.py b/.github/skills/anthropic-pdf/scripts/check_fillable_fields.py new file mode 100644 index 00000000..36dfb951 --- /dev/null +++ b/.github/skills/anthropic-pdf/scripts/check_fillable_fields.py @@ -0,0 +1,11 @@ +import sys +from pypdf import PdfReader + + + + +reader = PdfReader(sys.argv[1]) +if (reader.get_fields()): + print("This PDF has fillable form fields") +else: + print("This PDF does not have fillable form fields; you will need to visually determine where to enter data") diff --git a/.github/skills/anthropic-pdf/scripts/convert_pdf_to_images.py b/.github/skills/anthropic-pdf/scripts/convert_pdf_to_images.py new file mode 100644 index 00000000..7939cef5 --- /dev/null +++ b/.github/skills/anthropic-pdf/scripts/convert_pdf_to_images.py @@ -0,0 +1,33 @@ +import os +import sys + +from pdf2image import convert_from_path + + + + +def convert(pdf_path, output_dir, max_dim=1000): + images = convert_from_path(pdf_path, dpi=200) + + for i, image in enumerate(images): + width, height = image.size + if width > max_dim or height > max_dim: + scale_factor = min(max_dim / width, max_dim / height) + new_width = int(width * scale_factor) + new_height = int(height * scale_factor) + image = image.resize((new_width, new_height)) + + image_path = os.path.join(output_dir, f"page_{i+1}.png") + image.save(image_path) + print(f"Saved page {i+1} as {image_path} (size: {image.size})") + + print(f"Converted {len(images)} pages to PNG images") + + +if __name__ == "__main__": + if len(sys.argv) != 3: + print("Usage: convert_pdf_to_images.py [input pdf] [output directory]") + sys.exit(1) + pdf_path = sys.argv[1] + output_directory = sys.argv[2] + convert(pdf_path, output_directory) diff --git a/.github/skills/anthropic-pdf/scripts/create_validation_image.py b/.github/skills/anthropic-pdf/scripts/create_validation_image.py new file mode 100644 index 00000000..10eadd81 --- /dev/null +++ b/.github/skills/anthropic-pdf/scripts/create_validation_image.py @@ -0,0 +1,37 @@ +import json +import sys + +from PIL import Image, ImageDraw + + + + +def create_validation_image(page_number, fields_json_path, input_path, output_path): + with open(fields_json_path, 'r') as f: + data = json.load(f) + + img = Image.open(input_path) + draw = ImageDraw.Draw(img) + num_boxes = 0 + + for field in data["form_fields"]: + if field["page_number"] == page_number: + entry_box = field['entry_bounding_box'] + label_box = field['label_bounding_box'] + draw.rectangle(entry_box, outline='red', width=2) + draw.rectangle(label_box, outline='blue', width=2) + num_boxes += 2 + + img.save(output_path) + print(f"Created validation image at {output_path} with {num_boxes} bounding boxes") + + +if __name__ == "__main__": + if len(sys.argv) != 5: + print("Usage: create_validation_image.py [page number] [fields.json file] [input image path] [output image path]") + sys.exit(1) + page_number = int(sys.argv[1]) + fields_json_path = sys.argv[2] + input_image_path = sys.argv[3] + output_image_path = sys.argv[4] + create_validation_image(page_number, fields_json_path, input_image_path, output_image_path) diff --git a/.github/skills/anthropic-pdf/scripts/extract_form_field_info.py b/.github/skills/anthropic-pdf/scripts/extract_form_field_info.py new file mode 100644 index 00000000..64cd4703 --- /dev/null +++ b/.github/skills/anthropic-pdf/scripts/extract_form_field_info.py @@ -0,0 +1,122 @@ +import json +import sys + +from pypdf import PdfReader + + + + +def get_full_annotation_field_id(annotation): + components = [] + while annotation: + field_name = annotation.get('/T') + if field_name: + components.append(field_name) + annotation = annotation.get('/Parent') + return ".".join(reversed(components)) if components else None + + +def make_field_dict(field, field_id): + field_dict = {"field_id": field_id} + ft = field.get('/FT') + if ft == "/Tx": + field_dict["type"] = "text" + elif ft == "/Btn": + field_dict["type"] = "checkbox" + states = field.get("/_States_", []) + if len(states) == 2: + if "/Off" in states: + field_dict["checked_value"] = states[0] if states[0] != "/Off" else states[1] + field_dict["unchecked_value"] = "/Off" + else: + print(f"Unexpected state values for checkbox `${field_id}`. Its checked and unchecked values may not be correct; if you're trying to check it, visually verify the results.") + field_dict["checked_value"] = states[0] + field_dict["unchecked_value"] = states[1] + elif ft == "/Ch": + field_dict["type"] = "choice" + states = field.get("/_States_", []) + field_dict["choice_options"] = [{ + "value": state[0], + "text": state[1], + } for state in states] + else: + field_dict["type"] = f"unknown ({ft})" + return field_dict + + +def get_field_info(reader: PdfReader): + fields = reader.get_fields() + + field_info_by_id = {} + possible_radio_names = set() + + for field_id, field in fields.items(): + if field.get("/Kids"): + if field.get("/FT") == "/Btn": + possible_radio_names.add(field_id) + continue + field_info_by_id[field_id] = make_field_dict(field, field_id) + + + radio_fields_by_id = {} + + for page_index, page in enumerate(reader.pages): + annotations = page.get('/Annots', []) + for ann in annotations: + field_id = get_full_annotation_field_id(ann) + if field_id in field_info_by_id: + field_info_by_id[field_id]["page"] = page_index + 1 + field_info_by_id[field_id]["rect"] = ann.get('/Rect') + elif field_id in possible_radio_names: + try: + on_values = [v for v in ann["/AP"]["/N"] if v != "/Off"] + except KeyError: + continue + if len(on_values) == 1: + rect = ann.get("/Rect") + if field_id not in radio_fields_by_id: + radio_fields_by_id[field_id] = { + "field_id": field_id, + "type": "radio_group", + "page": page_index + 1, + "radio_options": [], + } + radio_fields_by_id[field_id]["radio_options"].append({ + "value": on_values[0], + "rect": rect, + }) + + fields_with_location = [] + for field_info in field_info_by_id.values(): + if "page" in field_info: + fields_with_location.append(field_info) + else: + print(f"Unable to determine location for field id: {field_info.get('field_id')}, ignoring") + + def sort_key(f): + if "radio_options" in f: + rect = f["radio_options"][0]["rect"] or [0, 0, 0, 0] + else: + rect = f.get("rect") or [0, 0, 0, 0] + adjusted_position = [-rect[1], rect[0]] + return [f.get("page"), adjusted_position] + + sorted_fields = fields_with_location + list(radio_fields_by_id.values()) + sorted_fields.sort(key=sort_key) + + return sorted_fields + + +def write_field_info(pdf_path: str, json_output_path: str): + reader = PdfReader(pdf_path) + field_info = get_field_info(reader) + with open(json_output_path, "w") as f: + json.dump(field_info, f, indent=2) + print(f"Wrote {len(field_info)} fields to {json_output_path}") + + +if __name__ == "__main__": + if len(sys.argv) != 3: + print("Usage: extract_form_field_info.py [input pdf] [output json]") + sys.exit(1) + write_field_info(sys.argv[1], sys.argv[2]) diff --git a/.github/skills/anthropic-pdf/scripts/extract_form_structure.py b/.github/skills/anthropic-pdf/scripts/extract_form_structure.py new file mode 100644 index 00000000..f219e7d5 --- /dev/null +++ b/.github/skills/anthropic-pdf/scripts/extract_form_structure.py @@ -0,0 +1,115 @@ +""" +Extract form structure from a non-fillable PDF. + +This script analyzes the PDF to find: +- Text labels with their exact coordinates +- Horizontal lines (row boundaries) +- Checkboxes (small rectangles) + +Output: A JSON file with the form structure that can be used to generate +accurate field coordinates for filling. + +Usage: python extract_form_structure.py +""" + +import json +import sys +import pdfplumber + + +def extract_form_structure(pdf_path): + structure = { + "pages": [], + "labels": [], + "lines": [], + "checkboxes": [], + "row_boundaries": [] + } + + with pdfplumber.open(pdf_path) as pdf: + for page_num, page in enumerate(pdf.pages, 1): + structure["pages"].append({ + "page_number": page_num, + "width": float(page.width), + "height": float(page.height) + }) + + words = page.extract_words() + for word in words: + structure["labels"].append({ + "page": page_num, + "text": word["text"], + "x0": round(float(word["x0"]), 1), + "top": round(float(word["top"]), 1), + "x1": round(float(word["x1"]), 1), + "bottom": round(float(word["bottom"]), 1) + }) + + for line in page.lines: + if abs(float(line["x1"]) - float(line["x0"])) > page.width * 0.5: + structure["lines"].append({ + "page": page_num, + "y": round(float(line["top"]), 1), + "x0": round(float(line["x0"]), 1), + "x1": round(float(line["x1"]), 1) + }) + + for rect in page.rects: + width = float(rect["x1"]) - float(rect["x0"]) + height = float(rect["bottom"]) - float(rect["top"]) + if 5 <= width <= 15 and 5 <= height <= 15 and abs(width - height) < 2: + structure["checkboxes"].append({ + "page": page_num, + "x0": round(float(rect["x0"]), 1), + "top": round(float(rect["top"]), 1), + "x1": round(float(rect["x1"]), 1), + "bottom": round(float(rect["bottom"]), 1), + "center_x": round((float(rect["x0"]) + float(rect["x1"])) / 2, 1), + "center_y": round((float(rect["top"]) + float(rect["bottom"])) / 2, 1) + }) + + lines_by_page = {} + for line in structure["lines"]: + page = line["page"] + if page not in lines_by_page: + lines_by_page[page] = [] + lines_by_page[page].append(line["y"]) + + for page, y_coords in lines_by_page.items(): + y_coords = sorted(set(y_coords)) + for i in range(len(y_coords) - 1): + structure["row_boundaries"].append({ + "page": page, + "row_top": y_coords[i], + "row_bottom": y_coords[i + 1], + "row_height": round(y_coords[i + 1] - y_coords[i], 1) + }) + + return structure + + +def main(): + if len(sys.argv) != 3: + print("Usage: extract_form_structure.py ") + sys.exit(1) + + pdf_path = sys.argv[1] + output_path = sys.argv[2] + + print(f"Extracting structure from {pdf_path}...") + structure = extract_form_structure(pdf_path) + + with open(output_path, "w") as f: + json.dump(structure, f, indent=2) + + print(f"Found:") + print(f" - {len(structure['pages'])} pages") + print(f" - {len(structure['labels'])} text labels") + print(f" - {len(structure['lines'])} horizontal lines") + print(f" - {len(structure['checkboxes'])} checkboxes") + print(f" - {len(structure['row_boundaries'])} row boundaries") + print(f"Saved to {output_path}") + + +if __name__ == "__main__": + main() diff --git a/.github/skills/anthropic-pdf/scripts/fill_fillable_fields.py b/.github/skills/anthropic-pdf/scripts/fill_fillable_fields.py new file mode 100644 index 00000000..51c2600f --- /dev/null +++ b/.github/skills/anthropic-pdf/scripts/fill_fillable_fields.py @@ -0,0 +1,98 @@ +import json +import sys + +from pypdf import PdfReader, PdfWriter + +from extract_form_field_info import get_field_info + + + + +def fill_pdf_fields(input_pdf_path: str, fields_json_path: str, output_pdf_path: str): + with open(fields_json_path) as f: + fields = json.load(f) + fields_by_page = {} + for field in fields: + if "value" in field: + field_id = field["field_id"] + page = field["page"] + if page not in fields_by_page: + fields_by_page[page] = {} + fields_by_page[page][field_id] = field["value"] + + reader = PdfReader(input_pdf_path) + + has_error = False + field_info = get_field_info(reader) + fields_by_ids = {f["field_id"]: f for f in field_info} + for field in fields: + existing_field = fields_by_ids.get(field["field_id"]) + if not existing_field: + has_error = True + print(f"ERROR: `{field['field_id']}` is not a valid field ID") + elif field["page"] != existing_field["page"]: + has_error = True + print(f"ERROR: Incorrect page number for `{field['field_id']}` (got {field['page']}, expected {existing_field['page']})") + else: + if "value" in field: + err = validation_error_for_field_value(existing_field, field["value"]) + if err: + print(err) + has_error = True + if has_error: + sys.exit(1) + + writer = PdfWriter(clone_from=reader) + for page, field_values in fields_by_page.items(): + writer.update_page_form_field_values(writer.pages[page - 1], field_values, auto_regenerate=False) + + writer.set_need_appearances_writer(True) + + with open(output_pdf_path, "wb") as f: + writer.write(f) + + +def validation_error_for_field_value(field_info, field_value): + field_type = field_info["type"] + field_id = field_info["field_id"] + if field_type == "checkbox": + checked_val = field_info["checked_value"] + unchecked_val = field_info["unchecked_value"] + if field_value != checked_val and field_value != unchecked_val: + return f'ERROR: Invalid value "{field_value}" for checkbox field "{field_id}". The checked value is "{checked_val}" and the unchecked value is "{unchecked_val}"' + elif field_type == "radio_group": + option_values = [opt["value"] for opt in field_info["radio_options"]] + if field_value not in option_values: + return f'ERROR: Invalid value "{field_value}" for radio group field "{field_id}". Valid values are: {option_values}' + elif field_type == "choice": + choice_values = [opt["value"] for opt in field_info["choice_options"]] + if field_value not in choice_values: + return f'ERROR: Invalid value "{field_value}" for choice field "{field_id}". Valid values are: {choice_values}' + return None + + +def monkeypatch_pydpf_method(): + from pypdf.generic import DictionaryObject + from pypdf.constants import FieldDictionaryAttributes + + original_get_inherited = DictionaryObject.get_inherited + + def patched_get_inherited(self, key: str, default = None): + result = original_get_inherited(self, key, default) + if key == FieldDictionaryAttributes.Opt: + if isinstance(result, list) and all(isinstance(v, list) and len(v) == 2 for v in result): + result = [r[0] for r in result] + return result + + DictionaryObject.get_inherited = patched_get_inherited + + +if __name__ == "__main__": + if len(sys.argv) != 4: + print("Usage: fill_fillable_fields.py [input pdf] [field_values.json] [output pdf]") + sys.exit(1) + monkeypatch_pydpf_method() + input_pdf = sys.argv[1] + fields_json = sys.argv[2] + output_pdf = sys.argv[3] + fill_pdf_fields(input_pdf, fields_json, output_pdf) diff --git a/.github/skills/anthropic-pdf/scripts/fill_pdf_form_with_annotations.py b/.github/skills/anthropic-pdf/scripts/fill_pdf_form_with_annotations.py new file mode 100644 index 00000000..b430069f --- /dev/null +++ b/.github/skills/anthropic-pdf/scripts/fill_pdf_form_with_annotations.py @@ -0,0 +1,107 @@ +import json +import sys + +from pypdf import PdfReader, PdfWriter +from pypdf.annotations import FreeText + + + + +def transform_from_image_coords(bbox, image_width, image_height, pdf_width, pdf_height): + x_scale = pdf_width / image_width + y_scale = pdf_height / image_height + + left = bbox[0] * x_scale + right = bbox[2] * x_scale + + top = pdf_height - (bbox[1] * y_scale) + bottom = pdf_height - (bbox[3] * y_scale) + + return left, bottom, right, top + + +def transform_from_pdf_coords(bbox, pdf_height): + left = bbox[0] + right = bbox[2] + + pypdf_top = pdf_height - bbox[1] + pypdf_bottom = pdf_height - bbox[3] + + return left, pypdf_bottom, right, pypdf_top + + +def fill_pdf_form(input_pdf_path, fields_json_path, output_pdf_path): + + with open(fields_json_path, "r") as f: + fields_data = json.load(f) + + reader = PdfReader(input_pdf_path) + writer = PdfWriter() + + writer.append(reader) + + pdf_dimensions = {} + for i, page in enumerate(reader.pages): + mediabox = page.mediabox + pdf_dimensions[i + 1] = [mediabox.width, mediabox.height] + + annotations = [] + for field in fields_data["form_fields"]: + page_num = field["page_number"] + + page_info = next(p for p in fields_data["pages"] if p["page_number"] == page_num) + pdf_width, pdf_height = pdf_dimensions[page_num] + + if "pdf_width" in page_info: + transformed_entry_box = transform_from_pdf_coords( + field["entry_bounding_box"], + float(pdf_height) + ) + else: + image_width = page_info["image_width"] + image_height = page_info["image_height"] + transformed_entry_box = transform_from_image_coords( + field["entry_bounding_box"], + image_width, image_height, + float(pdf_width), float(pdf_height) + ) + + if "entry_text" not in field or "text" not in field["entry_text"]: + continue + entry_text = field["entry_text"] + text = entry_text["text"] + if not text: + continue + + font_name = entry_text.get("font", "Arial") + font_size = str(entry_text.get("font_size", 14)) + "pt" + font_color = entry_text.get("font_color", "000000") + + annotation = FreeText( + text=text, + rect=transformed_entry_box, + font=font_name, + font_size=font_size, + font_color=font_color, + border_color=None, + background_color=None, + ) + annotations.append(annotation) + writer.add_annotation(page_number=page_num - 1, annotation=annotation) + + with open(output_pdf_path, "wb") as output: + writer.write(output) + + print(f"Successfully filled PDF form and saved to {output_pdf_path}") + print(f"Added {len(annotations)} text annotations") + + +if __name__ == "__main__": + if len(sys.argv) != 4: + print("Usage: fill_pdf_form_with_annotations.py [input pdf] [fields.json] [output pdf]") + sys.exit(1) + input_pdf = sys.argv[1] + fields_json = sys.argv[2] + output_pdf = sys.argv[3] + + fill_pdf_form(input_pdf, fields_json, output_pdf) diff --git a/.github/skills/anthropic-pptx/LICENSE.txt b/.github/skills/anthropic-pptx/LICENSE.txt new file mode 100644 index 00000000..c55ab422 --- /dev/null +++ b/.github/skills/anthropic-pptx/LICENSE.txt @@ -0,0 +1,30 @@ +© 2025 Anthropic, PBC. All rights reserved. + +LICENSE: Use of these materials (including all code, prompts, assets, files, +and other components of this Skill) is governed by your agreement with +Anthropic regarding use of Anthropic's services. If no separate agreement +exists, use is governed by Anthropic's Consumer Terms of Service or +Commercial Terms of Service, as applicable: +https://www.anthropic.com/legal/consumer-terms +https://www.anthropic.com/legal/commercial-terms +Your applicable agreement is referred to as the "Agreement." "Services" are +as defined in the Agreement. + +ADDITIONAL RESTRICTIONS: Notwithstanding anything in the Agreement to the +contrary, users may not: + +- Extract these materials from the Services or retain copies of these + materials outside the Services +- Reproduce or copy these materials, except for temporary copies created + automatically during authorized use of the Services +- Create derivative works based on these materials +- Distribute, sublicense, or transfer these materials to any third party +- Make, offer to sell, sell, or import any inventions embodied in these + materials +- Reverse engineer, decompile, or disassemble these materials + +The receipt, viewing, or possession of these materials does not convey or +imply any license or right beyond those expressly granted above. + +Anthropic retains all right, title, and interest in these materials, +including all copyrights, patents, and other intellectual property rights. diff --git a/.github/skills/anthropic-pptx/SKILL.md b/.github/skills/anthropic-pptx/SKILL.md new file mode 100644 index 00000000..017dda04 --- /dev/null +++ b/.github/skills/anthropic-pptx/SKILL.md @@ -0,0 +1,238 @@ +--- +name: anthropic-pptx +description: "Use this skill any time a .pptx or .potx file is involved in any way — as input, output, or both. This includes: creating slide decks, pitch decks, or presentations; reading, parsing, or extracting text from any .pptx or .potx file (even if the extracted content will be used elsewhere, like in an email or summary); editing, modifying, or updating existing presentations; combining or splitting slide files; working with templates (.potx), layouts, speaker notes, or comments. Trigger whenever the user mentions \"deck,\" \"slides,\" \"presentation,\" or references a .pptx or .potx filename, regardless of what they plan to do with the content afterward. If a .pptx or .potx file needs to be opened, created, or touched, use this skill." +license: Proprietary. LICENSE.txt has complete terms +--- + +# PPTX creation, editing, and analysis + +A `.pptx` is a ZIP archive of XML files. Choose your approach by task: + +| Task | Approach | +|---|---| +| **Create** a new deck | Write a `pptxgenjs` script — see gotchas below | +| **Edit** an existing deck, or build from a template | unzip → edit `ppt/slides/slideN.xml` → zip | +| **Read** content | `markitdown deck.pptx` (one block per slide under `` markers); visual grid: `python scripts/thumbnail.py deck.pptx` | + +## Scripts + +Paths are relative to this skill's directory. Everything else is plain Python, `node`, or shell. + +| Script | What it does | +|---|---| +| `scripts/thumbnail.py deck.pptx [prefix]` | Labeled grid of every slide, for picking template layouts. `.pptx` only. Pass `prefix` — it defaults to `thumbnails`, which overwrites the grids of any other deck done in the same directory | +| `scripts/add_slide.py unpacked/ slide2.xml [--after slideN.xml]` | Duplicate a slide (or a `slideLayoutN.xml`) with all the package bookkeeping. Also takes a `.pptx` directly with `-o out.pptx` | +| `scripts/clean.py unpacked/` | Delete slides, media, and rels no longer referenced. Run **after** `` is final | +| `scripts/office/validate.py deck.pptx [--original src.pptx]` | Schema, relationship, content-type, chart and slide checks; each failure names its fix. Pass `--original` for any template-derived deck — it baselines the schema checks against the template, so the template's own XSD errors don't read as yours | +| `scripts/office/soffice.py --headless --convert-to pdf deck.pptx` | LibreOffice wrapper — bare `soffice` hangs in this sandbox | + +## Creating with pptxgenjs — gotchas + +`pptxgenjs` is preinstalled — do not run `npm install` first; write the script and `require('pptxgenjs')` directly. Only if that require fails: `npm install pptxgenjs`. The model knows the API; these are the footguns: + +- **Set `pres.layout` before adding slides.** The default canvas is `LAYOUT_16x9` = **10" × 5.625"**, not 13.3" wide. Coordinates past the edge are written, not clamped — the shape just isn't on the slide. (`LAYOUT_WIDE` is 13.3" × 7.5".) +- **Hex colors: never `#`, never 8 digits.** `color: "FF0000"`. Both `"#FF0000"` and alpha baked into the hex (`"00000020"`) **corrupt the file**. For translucency: `transparency: 0-100` on fills and images, `opacity: 0.0-1.0` on shadows — each is silently ignored on the other. +- **pptxgenjs mutates option objects in place** (converts values to EMU on first use). Never share one `shadow`/options object across two `add*` calls — build a fresh object each time. +- **Shadow `offset` must be ≥ 0** — a negative offset corrupts the file. To cast a shadow upward, use `angle: 270` with a positive offset. +- **`letterSpacing` is silently ignored** — the real option is `charSpacing`. +- **Lists:** `bullet: true` on each item, never a literal `•` (renders double bullets). Set `breakLine: true` on every array item except the last. Space bulleted paragraphs with `paraSpaceAfter`, not `lineSpacing` (huge gaps). +- **One `new pptxgen()` per output file** — never reuse an instance. +- **`rectRadius` only works on `ROUNDED_RECTANGLE`**, not `RECTANGLE`. +- **Gradient fills aren't supported** — use a gradient image as the background instead. +- **Text boxes have built-in internal padding** — set `margin: 0` whenever text must align with a shape, line, or icon at the same x. +- **Speaker notes go in `slide.addNotes("...")`** (plain text, once per slide), never in a text box on the slide. +- **Keep charts native.** Use `addChart()` for everything PowerPoint can chart (pass an array of `{type, data, options}` for combos). For PowerPoint-native features the library doesn't expose (trendlines, error bars), compute the extra series yourself or post-process the generated OOXML — do not fall back to a rendered image. Only chart types PowerPoint has no native form for (Sankey, network, chord) go in as images. +- **Default charts render bare** — no title, no data labels, dated palette. Set `showTitle` + `title`, `showValue: true` + `dataLabelPosition`, `chartColors: [...]` from your palette, and quiet the frame (`catAxisLabelColor`/`valAxisLabelColor`, `valGridLine: { color, size }`, `catGridLine: { style: "none" }`, `showLegend: false` for a single series). +- **On a stacked bar or column chart, `dataLabelPosition` must be `ctr`, `inEnd`, or `inBase`.** `outEnd` **corrupts the file**. +- **A combo series using `secondaryValAxis`/`secondaryCatAxis` needs both `valAxes` and `catAxes` on the chart options, two entries each.** Without them pptxgenjs writes axis *ids* it never declares, and PowerPoint **discards that chart** and reports the file as corrupt. Supplying only `valAxes` is not enough. +- **After `writeFile()`, run `python scripts/office/validate.py deck.pptx`.** It reports the two chart faults above and the slide-XML defects PowerPoint refuses, and names the fix for each. Fix them in your generator, not by hand-editing the packed XML. +- **Never reorder the children of ``.** pptxgenjs writes `` right after `` and points both masters at one theme part. PowerPoint reads that happily — move the element and the same deck becomes unopenable. +- **Icons:** render `react-icons` to SVG (`ReactDOMServer.renderToStaticMarkup`), rasterize with `sharp` at ≥256px, and insert via `addImage({ data: "image/png;base64," + buf.toString("base64") })` — the `image/png;base64,` prefix is required (`react-icons`, `react`, `react-dom`, and `sharp` are preinstalled — `npm install react-icons react react-dom sharp` only if a require fails). + +## Editing existing decks and templates + +Pick layouts first: `python scripts/thumbnail.py template.pptx template-thumbs` writes a labeled grid of every slide and prints the file(s) it created — `template-thumbs.jpg`, split into `template-thumbs-N.jpg` past 12 slides. **Always pass that second argument, named after the deck.** It defaults to `thumbnails`, so two decks thumbnailed in one directory silently overwrite each other's grids — the first deck's are simply gone (template analysis only — visual QA needs the full-resolution renders from [Converting to Images](#converting-to-images); it only accepts `.pptx`, so copy a `.potx` to a `.pptx` name first). Use it with `markitdown` to map each content section onto a template slide, and vary the layouts — don't put every section on the same title-and-bullets slide. + +```bash +python3 -c "import sys,zipfile; zipfile.ZipFile(sys.argv[1]).extractall('unpacked')" deck.pptx +python scripts/add_slide.py unpacked/ slide2.xml --after slide2.xml # duplicate a slide (or slideLayoutN.xml); prints the new slide's path +# reorder / delete slides = edit in ppt/presentation.xml +python scripts/clean.py unpacked/ # after deletions: removes orphaned slides, media, rels +# edit slide content in ppt/slides/slideN.xml +(cd unpacked && rm -f ../out.pptx && zip -Xr ../out.pptx .) # zip from INSIDE the dir; rm first or deleted parts survive +python scripts/office/validate.py out.pptx --original deck.pptx +``` + +- **Do all structural work — add, delete, reorder — before editing any slide's content.** `add_slide.py` copies a slide file verbatim, so duplicating after you edit clones the edited content; and `clean.py` deletes any slide missing from ``, including one you just wrote. +- **Never copy a slide file by hand** — `add_slide.py` does every registration a new slide needs and reports what it made (`Created ppt/slides/slide17.xml from slide2.xml`). It also works directly on a file: `add_slide.py deck.pptx slide2.xml -o out.pptx` — **pass `-o`, or it rewrites the input deck in place.** A duplicated slide still *references* its source's chart/SmartArt/embedded-object parts rather than cloning them, so editing one slide's chart changes the other's. +- **If you use `python-pptx`**, three things it won't do: duplicate a slide (its only entry point is `add_slide(layout)`), preserve formatting through `text_frame.text = "..."` (that collapses the paragraph to a single unstyled run — assign `run.text` instead), or read the SVG/EMF most template art uses (`add_picture` raises `UnidentifiedImageError`). +- Legacy `.ppt` must be converted first: `python scripts/office/soffice.py --headless --convert-to pptx file.ppt`. `.potx` templates unpack and pack identically — keep the `.potx` extension on the output. +- To reuse a template icon or image, duplicate a slide or layout that already contains it. + +When filling in a template: + +- If you script an XML transform, parse with `defusedxml.minidom` — round-tripping OOXML through `xml.etree.ElementTree` rewrites namespace prefixes and corrupts the deck. +- **Template slots ≠ source items.** If the template shows 4 team members and you have 3, delete the 4th member's entire group (image + text boxes), not just its text — then check for orphaned visuals in QA. +- One `` per list item — never concatenate items into a single paragraph. Copy the sibling `` to preserve spacing, and put `b="1"` on the `` of titles, section headers, and inline labels (`Status:`, `Owner:`). +- Let bullets inherit from the layout; only add ``, `` (numbered), or `` to override — never a literal `•` in the text. +- Text with leading or trailing spaces needs `xml:space="preserve"` on its ``. + +## Design Ideas + +**Don't create boring slides.** Plain bullets on a white background won't impress anyone. Consider ideas from this list for each slide. + +### Before Starting + +- **Pick a bold, content-informed color palette**: The palette should feel designed for THIS topic. If swapping your colors into a completely different presentation would still "work," you haven't made specific enough choices. +- **Dominance over equality**: One color should dominate (60-70% visual weight), with 1-2 supporting tones and one sharp accent. Never give all colors equal weight. +- **Dark/light contrast**: Dark backgrounds for title + conclusion slides, light for content ("sandwich" structure). Or commit to dark throughout for a premium feel. +- **Commit to a visual motif**: Pick ONE distinctive element and repeat it — rounded image frames, icons in colored circles. Carry it across every slide. **Do not use a color bar or accent stripe as your motif** (see Avoid list). + +### Color Palettes + +Choose colors that match your topic — don't default to generic blue. Use these palettes as inspiration: + +| Theme | Primary | Secondary | Accent | +|-------|---------|-----------|--------| +| **Midnight Executive** | `1E2761` (navy) | `CADCFC` (ice blue) | `FFFFFF` (white) | +| **Forest & Moss** | `2C5F2D` (forest) | `97BC62` (moss) | `F5F5F5` (cream) | +| **Coral Energy** | `F96167` (coral) | `F9E795` (gold) | `2F3C7E` (navy) | +| **Warm Terracotta** | `B85042` (terracotta) | `E7E8D1` (sand) | `A7BEAE` (sage) | +| **Ocean Gradient** | `065A82` (deep blue) | `1C7293` (teal) | `21295C` (midnight) | +| **Charcoal Minimal** | `36454F` (charcoal) | `F2F2F2` (off-white) | `212121` (black) | +| **Teal Trust** | `028090` (teal) | `00A896` (seafoam) | `02C39A` (mint) | +| **Berry & Cream** | `6D2E46` (berry) | `A26769` (dusty rose) | `ECE2D0` (cream) | +| **Sage Calm** | `84B59F` (sage) | `69A297` (eucalyptus) | `50808E` (slate) | +| **Cherry Bold** | `990011` (cherry) | `FCF6F5` (off-white) | `2F3C7E` (navy) | + +### For Each Slide + +**Every slide needs a visual element** — image, chart, icon, or shape. Text-only slides are forgettable. + +**Layout options:** +- Two-column (text left, illustration on right) +- Icon + text rows (icon in colored circle, bold header, description below) +- 2x2 or 2x3 grid (image on one side, grid of content blocks on other) +- Half-bleed image (full left or right side) with content overlay + +**Data display:** +- Large stat callouts (big numbers 60-72pt with small labels below) +- Comparison columns (before/after, pros/cons, side-by-side options) +- Timeline or process flow (numbered steps, arrows) + +**Visual polish:** +- Icons in small colored circles next to section headers +- Italic accent text for key stats or taglines + +### Typography + +**Font names you write into the .pptx are rendered by the user's PowerPoint, not by this environment.** Your visual QA renders via LibreOffice, which substitutes fonts it doesn't have — and for some fonts the substitute has different widths, so your QA preview can show text overflow (or fit) that the real deck won't have. To keep your QA trustworthy: + +- **Safe fonts** (render true-to-width in QA *and* ship with Office): **Arial, Calibri, Cambria, Times New Roman, Courier New, Bookman Old Style, Century Schoolbook**. Use these for body text and anything where fit matters. +- **Headers with personality at zero QA risk**: pair a safe-list serif header (Cambria, Bookman Old Style, Century Schoolbook) with a safe-list sans body (Calibri or Arial). You get visual contrast without giving up reliable overflow checks. +- **If the user asks for a font outside the safe list** (e.g. Georgia or Trebuchet MS): use it where the user asked, but size those containers with extra slack (~10%) and don't trust QA text-fit on those elements — the preview of that font is approximate. If the user hasn't specified, prefer safe-list fonts for body text. +- **QA-unreliable fonts** (substitute has different widths — overflow checks can be wrong): Georgia, Trebuchet MS, Impact, Arial Black, Garamond, Consolas, Palatino Linotype. Calibri Light substitution varies by environment; treat as QA-unreliable. Fine for titles/accents with slack; don't trust QA text-fit on these. +- **Never default to Aptos** — Office's post-2023 default has no metric-compatible substitute here *and* is missing from older Office installs, so it's unreliable on both ends. + +| Element | Size | +|---------|------| +| Slide title | 36-44pt bold | +| Section header | 20-24pt bold | +| Body text | 14-16pt | +| Captions | 10-12pt muted | + +### Spacing + +- 0.5" minimum margins +- 0.3-0.5" between content blocks +- Leave breathing room—don't fill every inch + +### Avoid (Common Mistakes) + +- **Don't repeat the same layout** — vary columns, cards, and callouts across slides +- **Don't center body text** — left-align paragraphs and lists; center only titles +- **Don't skimp on size contrast** — titles need 36pt+ to stand out from 14-16pt body +- **Don't default to blue** — pick colors that reflect the specific topic +- **Don't mix spacing randomly** — choose 0.3" or 0.5" gaps and use consistently +- **Don't style one slide and leave the rest plain** — commit fully or keep it simple throughout +- **Don't create text-only slides** — add images, icons, charts, or visual elements; avoid plain title + bullets +- **Don't forget text box padding** — when aligning lines or shapes with text edges, set `margin: 0` on the text box or offset the shape to account for padding +- **Don't use low-contrast elements** — icons AND text need strong contrast against the background; avoid light text on light backgrounds or dark text on dark backgrounds +- **NEVER use accent lines under titles** — these are a hallmark of AI-generated slides; use whitespace or background color instead +- **NEVER add decorative color bars or accent stripes** — this includes: header/footer bars spanning the slide width, vertical sidebar stripes down one edge of the slide, thin accent stripes along one edge of a card or content block, and "single-side borders" on rectangles. These read as AI-generated filler. If you want to set a card apart, use a subtle background tint, a drop shadow, or an icon — not an edge stripe. +- **Don't default to cream/beige backgrounds** — when no background is specified, use white (`FFFFFF`) or the user's brand palette; avoid warm-neutral defaults like `F5F5DC`, `FAF0E6`, `FAEBD7`, `FFF8E1` +- **Don't ship text that overflows its shape** — if text doesn't fit, reduce font size, split across slides, or enlarge the container; never leave content cut off or spilling past bounds + +## QA (Required) + +Your first render usually has a few real issues — overlaps, overflow, misalignment. Find and fix those, re-render only the slides you changed, and stop. + +### Content QA + +```bash +markitdown output.pptx +``` + +Check for missing content, typos, wrong order. + +**When using templates, check for leftover placeholder text:** + +```bash +markitdown output.pptx | grep -iE "\bx{3,}\b|lorem|ipsum|\bTODO|\[insert|this.*(page|slide).*layout" +``` + +If grep returns results, fix them before declaring success. + +### File QA (required) + +```bash +python scripts/office/validate.py output.pptx # built from scratch +python scripts/office/validate.py output.pptx --original src.pptx # built from a template +``` + +**If the deck came from a template, always pass `--original`.** A template may itself +contain parts the XSD rejects, so a bare run can report failures you never caused — and +a genuine regression can hide among them. `--original` baselines +the schema and slide checks against the template, suppressing errors it already had. +The structural checks — relationships, content types, charts — ignore `--original` and +report template-inherited problems either way, so read those on their own merits. + +pptxgenjs emits chart XML PowerPoint refuses to open, and every other tool +accepts: python-pptx opens those decks, LibreOffice renders them, the XSD +passes them. Every failure names its fix. Fix it in the generator and rebuild. + +### Visual QA + +Convert the slides to images (see [Converting to Images](#converting-to-images)) and inspect every one. After staring at the generating code you tend to see what you expect rather than what rendered, so look at the images fresh (a subagent works well for this if you have one). User-visible defects to look for: + +- **Text overflow or text cut off at a box or slide boundary — check this first.** It is the most common defect and always user-visible. (For a font the previewer renders unreliably per Typography, the preview is approximate: trust the ~10% slack you left, not its apparent fit.) +- Overlapping elements (text through shapes, lines through words, stacked elements) +- Source citations or footers colliding with content above +- Elements too close (< 0.3" gaps) or cards/sections nearly touching +- Uneven gaps (large empty area in one place, cramped in another) +- Insufficient margin from slide edges (< 0.5") +- Columns or similar elements not aligned consistently +- Low-contrast text (e.g., light gray text on cream-colored background) +- Template decoration mispositioned after text replacement — e.g., a title underline positioned for one line, but the replaced title wrapped to two +- Low-contrast icons (e.g., dark icons on dark backgrounds without a contrasting circle) +- Text boxes too narrow causing excessive wrapping +- Leftover placeholder content + +## Converting to Images + +Convert presentations to individual slide images for visual inspection: + +```bash +python scripts/office/soffice.py --headless --convert-to pdf output.pptx +rm -f slide-*.jpg +pdftoppm -jpeg -r 150 output.pdf slide +ls -1 "$PWD"/slide-*.jpg +``` + +**Pass the absolute paths printed above directly to the view tool.** The `rm` clears stale images from prior runs. `pdftoppm` zero-pads based on page count: `slide-1.jpg` for decks under 10 pages, `slide-01.jpg` for 10-99, `slide-001.jpg` for 100+. + +**After fixes, rerun all four commands above** — the PDF must be regenerated from the edited `.pptx` before `pdftoppm` can reflect your changes. + +## Dependencies + +`pptxgenjs` (npm, preinstalled — install only if `require('pptxgenjs')` fails) · `markitdown[pptx]`, `Pillow`, `defusedxml`, `lxml` (pip — text dump, thumbnail, clean, validate) · LibreOffice (`soffice`, auto-configured for sandboxed environments via `scripts/office/soffice.py`) · `pdftoppm` (Poppler) diff --git a/.github/skills/anthropic-pptx/scripts/__init__.py b/.github/skills/anthropic-pptx/scripts/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/.github/skills/anthropic-pptx/scripts/add_slide.py b/.github/skills/anthropic-pptx/scripts/add_slide.py new file mode 100644 index 00000000..f013ea94 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/add_slide.py @@ -0,0 +1,367 @@ +"""Add a slide to a PPTX: duplicate an existing slide or instantiate a layout. + +Does all of the package bookkeeping, so the deck stays valid: + - writes the new ppt/slides/slideN.xml (and its .rels, minus any + notesSlide reference, so the source's speaker notes aren't shared) + - registers it in [Content_Types].xml + - adds a slide relationship with a fresh rId to presentation.xml.rels + - inserts with a fresh id into + — at the end, or after --after SLIDE + +Works on an unpacked directory (during an editing session) or directly on a +.pptx/.potx file (extracted to a temp dir, then rezipped atomically; the +temp dir is discarded, so unpack the output if you still need to edit the +new slide's content). + +Usage: + python add_slide.py unpacked/ slide2.xml # duplicate slide2 + python add_slide.py unpacked/ slideLayout3.xml # new slide from a layout + python add_slide.py unpacked/ slide2.xml --after slide2.xml + python add_slide.py deck.pptx slide2.xml # rewrite deck.pptx in place + python add_slide.py deck.pptx slide2.xml -o out.pptx + +A duplicated slide still holds the source's content: edit ppt/slides/slideN.xml +(printed on success) to change it. To list layouts: ls /ppt/slideLayouts/ +""" + +import argparse +import re +import shutil +import sys +from typing import NoReturn +import tempfile +import zipfile +from pathlib import Path + +from office.helpers import rezip, safe_extract + +MINIMAL_SLIDE_XML = ''' + + + + + + + + + + + + + + + + + + + + + +''' + +SHARED_PART_TYPES = ("chart", "diagramData", "oleObject", "package") + +NOTES_SLIDE_TYPE_RE = re.compile(r"""Type=["'][^"']*/relationships/notesSlide["']""") +RELATIONSHIP_RE = re.compile(r"]*?(?:/>|>.*?)", re.DOTALL) + +SLIDE_ID_MIN = 256 +SLIDE_ID_MAX = 2147483647 + + +def _die(msg: str) -> NoReturn: + print(f"Error: {msg}", file=sys.stderr) + sys.exit(1) + + +def get_next_slide_number(slides_dir: Path) -> int: + existing = [int(m.group(1)) for f in slides_dir.glob("slide*.xml") + if (m := re.match(r"slide(\d+)\.xml", f.name))] + return max(existing) + 1 if existing else 1 + + +def parse_source(source: str) -> tuple[str, str | None]: + if source.startswith("slideLayout") and source.endswith(".xml"): + return ("layout", source) + + return ("slide", None) + + +def create_slide_from_layout(unpacked_dir: Path, layout_file: str, after: str | None = None) -> str: + slides_dir = unpacked_dir / "ppt" / "slides" + rels_dir = slides_dir / "_rels" + layout_path = unpacked_dir / "ppt" / "slideLayouts" / layout_file + + if not layout_path.exists(): + _die(f"{layout_path} not found") + + next_num = get_next_slide_number(slides_dir) + dest = f"slide{next_num}.xml" + after_rid = _precheck_registration(unpacked_dir, after, dest) + slides_dir.mkdir(parents=True, exist_ok=True) + + (slides_dir / dest).write_text(MINIMAL_SLIDE_XML, encoding="utf-8") + + rels_dir.mkdir(exist_ok=True) + rels_xml = f''' + + +''' + (rels_dir / f"{dest}.rels").write_text(rels_xml, encoding="utf-8") + + _register_slide(unpacked_dir, dest, layout_file, after_rid) + return dest + + +def duplicate_slide(unpacked_dir: Path, source: str, after: str | None = None) -> str: + slides_dir = unpacked_dir / "ppt" / "slides" + rels_dir = slides_dir / "_rels" + source_slide = slides_dir / source + + if not source_slide.exists(): + _die(f"{source_slide} not found") + + next_num = get_next_slide_number(slides_dir) + dest = f"slide{next_num}.xml" + after_rid = _precheck_registration(unpacked_dir, after, dest) + + shutil.copy2(source_slide, slides_dir / dest) + + source_rels = rels_dir / f"{source}.rels" + shared_parts: list[str] = [] + if source_rels.exists(): + dest_rels = rels_dir / f"{dest}.rels" + shutil.copy2(source_rels, dest_rels) + rels_content = dest_rels.read_text(encoding="utf-8") + rels_content = RELATIONSHIP_RE.sub( + lambda m: "" if NOTES_SLIDE_TYPE_RE.search(m.group(0)) else m.group(0), + rels_content, + ) + dest_rels.write_text(rels_content, encoding="utf-8") + shared_parts = sorted({ + t for t in re.findall(r'Type="[^"]*/relationships/(\w+)"', rels_content) + if t in SHARED_PART_TYPES + }) + + _register_slide(unpacked_dir, dest, source, after_rid) + if shared_parts: + print( + f"Note: {dest} shares its {', '.join(shared_parts)} part(s) with {source} " + f"(they are referenced, not copied) — editing those parts changes both slides" + ) + return dest + + +def _precheck_registration(unpacked_dir: Path, after: str | None, dest: str) -> str | None: + pres_path = unpacked_dir / "ppt" / "presentation.xml" + if not pres_path.exists(): + _die(f"{pres_path} not found — is this an unpacked PPTX?") + xml = pres_path.read_text(encoding="utf-8") + + has_slot = ( + "" in xml + or re.search(r"", xml) + or "" in xml + ) + if not has_slot: + _die("presentation.xml has no (or to anchor a new one)") + + stale = [] + content_types = unpacked_dir / "[Content_Types].xml" + if content_types.exists() and f'PartName="/ppt/slides/{dest}"' in content_types.read_text(encoding="utf-8"): + stale.append("[Content_Types].xml") + pres_rels = unpacked_dir / "ppt" / "_rels" / "presentation.xml.rels" + if pres_rels.exists() and _find_slide_relationship( + pres_rels.read_text(encoding="utf-8"), dest + ): + stale.append("presentation.xml.rels") + if stale: + _die( + f"{dest} is still registered in {' and '.join(stale)} but absent from ppt/slides/ — " + f"run clean.py first" + ) + + if not after: + return None + after_rid = _rid_for_slide(unpacked_dir, after) + if not re.search(rf']*r:id="{re.escape(after_rid)}"[^>]*>', xml): + _die(f"{after} ({after_rid}) is not listed in ") + return after_rid + + +def _register_slide(unpacked_dir: Path, dest: str, source_desc: str, after_rid: str | None) -> None: + _add_to_content_types(unpacked_dir, dest) + rid = _add_to_presentation_rels(unpacked_dir, dest) + slide_id = _get_next_slide_id(unpacked_dir) + pos, total = _insert_into_sld_id_lst(unpacked_dir, slide_id, rid, after_rid) + + print(f"Created ppt/slides/{dest} from {source_desc}") + print( + f'Inserted into ' + f"at position {pos} of {total}" + ) + + +def _add_to_content_types(unpacked_dir: Path, dest: str) -> None: + content_types_path = unpacked_dir / "[Content_Types].xml" + content_types = content_types_path.read_text(encoding="utf-8") + + new_override = f'' + + if f'PartName="/ppt/slides/{dest}"' not in content_types: + content_types = content_types.replace("", f" {new_override}\n") + content_types_path.write_text(content_types, encoding="utf-8") + + +def _add_to_presentation_rels(unpacked_dir: Path, dest: str) -> str: + pres_rels_path = unpacked_dir / "ppt" / "_rels" / "presentation.xml.rels" + pres_rels = pres_rels_path.read_text(encoding="utf-8") + + existing = _find_slide_relationship(pres_rels, dest) + if existing: + return existing + + pres_xml = (unpacked_dir / "ppt" / "presentation.xml").read_text(encoding="utf-8") + used = {int(n) for n in re.findall(r'\bId="rId(\d+)"', pres_rels)} + used |= {int(n) for n in re.findall(r'\br:id="rId(\d+)"', pres_xml)} + rid = f"rId{max(used) + 1 if used else 1}" + + new_rel = f'' + pres_rels = pres_rels.replace("", f" {new_rel}\n") + pres_rels_path.write_text(pres_rels, encoding="utf-8") + + return rid + + +def _find_slide_relationship(pres_rels: str, slide_name: str) -> str | None: + for m in re.finditer(r"]*>", pres_rels): + element = m.group(0) + if re.search(rf'Target="(?:/ppt/)?slides/{re.escape(slide_name)}"', element): + id_match = re.search(r'\bId="([^"]+)"', element) + if id_match: + return id_match.group(1) + return None + + +def _get_next_slide_id(unpacked_dir: Path) -> int: + pres_content = (unpacked_dir / "ppt" / "presentation.xml").read_text(encoding="utf-8") + used = {int(m) for m in re.findall(r']*\bid="(\d+)"', pres_content)} + + candidate = max((i for i in used if i >= SLIDE_ID_MIN), default=SLIDE_ID_MIN - 1) + 1 + if candidate <= SLIDE_ID_MAX and candidate not in used: + return candidate + for i in range(SLIDE_ID_MIN, SLIDE_ID_MAX + 1): + if i not in used: + return i + _die("no slide id available in [256, 2147483647] — the deck is full") + + +def _insert_into_sld_id_lst( + unpacked_dir: Path, slide_id: int, rid: str, after_rid: str | None = None +) -> tuple[int, int]: + pres_path = unpacked_dir / "ppt" / "presentation.xml" + xml = pres_path.read_text(encoding="utf-8") + entry = f'' + + if f'r:id="{rid}"' in xml: + _die(f"presentation.xml already references {rid}; refusing to add a duplicate") + + if after_rid: + open_tag = re.search(rf']*r:id="{re.escape(after_rid)}"[^>]*>', xml) + if not open_tag: + _die(f"{after_rid} is not listed in ") + end = open_tag.end() + if not open_tag.group(0).endswith("/>"): + close = xml.find("", end) + if close == -1: + _die(f"unclosed for {after_rid} in presentation.xml") + end = close + len("") + xml = xml[:end] + entry + xml[end:] + elif "" in xml: + xml = xml.replace("", f"{entry}", 1) + elif re.search(r"", xml): + xml = re.sub(r"", f"{entry}", xml, count=1) + elif "" in xml: + xml = xml.replace( + "", f"{entry}", 1 + ) + else: + _die("presentation.xml has no (or to anchor a new one)") + + pres_path.write_text(xml, encoding="utf-8") + + lst = re.search(r"(.*)", xml, re.DOTALL) + entries = re.findall(r"]*>", lst.group(1)) if lst else [] + position = next( + (i for i, e in enumerate(entries, 1) if f'r:id="{rid}"' in e), len(entries) + ) + return position, len(entries) + + +def _rid_for_slide(unpacked_dir: Path, slide_name: str) -> str: + pres_rels_path = unpacked_dir / "ppt" / "_rels" / "presentation.xml.rels" + rid = _find_slide_relationship(pres_rels_path.read_text(encoding="utf-8"), slide_name) + if not rid: + _die(f"{slide_name} has no relationship in presentation.xml.rels") + return rid + + +def add_slide(unpacked_dir: Path, source: str, after: str | None = None) -> str: + source_type, layout_file = parse_source(source) + if source_type == "layout" and layout_file is not None: + return create_slide_from_layout(unpacked_dir, layout_file, after) + return duplicate_slide(unpacked_dir, source, after) + + +def add_slide_to_package( + package: Path, source: str, after: str | None = None, output: Path | None = None +) -> str: + out = output or package + with tempfile.TemporaryDirectory() as tmp: + tmp_path = Path(tmp) + with zipfile.ZipFile(package) as zf: + safe_extract(zf, tmp_path) + dest = add_slide(tmp_path, source, after) + rezip(tmp_path, out) + print(f"Wrote {out} — the new slide is ppt/slides/{dest} inside it (unpack to edit its content)") + return dest + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Add a slide to a PPTX: duplicate a slide or instantiate a layout. " + "Registers content types, relationships, and ." + ) + parser.add_argument("target", help="Unpacked PPTX directory OR a .pptx/.potx file") + parser.add_argument( + "source", + help="slideN.xml to duplicate, or slideLayoutN.xml to create from a layout " + "(list layouts with: ls /ppt/slideLayouts/)", + ) + parser.add_argument( + "--after", + metavar="SLIDE", + help="insert after this slide, e.g. slide2.xml (default: append at the end)", + ) + parser.add_argument( + "-o", + "--output", + help="output file (only with a .pptx/.potx target; default: rewrite the input in place)", + ) + args = parser.parse_args() + + target = Path(args.target) + if target.is_dir(): + if args.output: + parser.error("--output is only valid for .pptx/.potx input; a directory is modified in place") + add_slide(target, args.source, args.after) + elif target.is_file() and target.suffix.lower() in (".pptx", ".potx"): + try: + add_slide_to_package(target, args.source, args.after, Path(args.output) if args.output else None) + except (OSError, ValueError, zipfile.BadZipFile) as e: + _die(str(e)) + else: + _die(f"{target} is neither a directory nor a .pptx/.potx file") + + +if __name__ == "__main__": + main() diff --git a/.github/skills/anthropic-pptx/scripts/clean.py b/.github/skills/anthropic-pptx/scripts/clean.py new file mode 100644 index 00000000..551dd231 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/clean.py @@ -0,0 +1,309 @@ +"""Remove unreferenced files from an unpacked PPTX directory. + +Usage: python clean.py + +Example: + python clean.py unpacked/ + +This script removes: +- Orphaned slides (not in sldIdLst) and their relationships +- [trash] directory (unreferenced files) +- Orphaned .rels files for deleted resources +- Unreferenced media, embeddings, charts, diagrams, drawings, ink files +- Unreferenced theme files +- Unreferenced notes slides +- Content-Type overrides for deleted files +""" + +import posixpath +import re +import sys +from pathlib import Path + +import defusedxml.minidom + +from office.helpers import SLIDE_REL_TYPE, opc_target, rels_source_part + + +def _slide_rids(pres_rels_path: Path, unpacked_dir: Path) -> dict[str, str]: + source_part = rels_source_part(pres_rels_path, unpacked_dir) + rels_dom = defusedxml.minidom.parse(str(pres_rels_path)) + + rids: dict[str, str] = {} + for rel in rels_dom.getElementsByTagName("Relationship"): + if rel.getAttribute("Type") != SLIDE_REL_TYPE: + continue + part = opc_target( + rel.getAttribute("Target"), source_part, rel.getAttribute("TargetMode") + ) + if part is not None: + rids[rel.getAttribute("Id")] = part + return rids + + +def get_slides_in_sldidlst(unpacked_dir: Path) -> set[str]: + pres_path = unpacked_dir / "ppt" / "presentation.xml" + pres_rels_path = unpacked_dir / "ppt" / "_rels" / "presentation.xml.rels" + + if not pres_path.exists() or not pres_rels_path.exists(): + return set() + + rid_to_slide = _slide_rids(pres_rels_path, unpacked_dir) + + pres_content = pres_path.read_text(encoding="utf-8") + referenced_rids = set(re.findall(r']*r:id="([^"]+)"', pres_content)) + + return { + posixpath.basename(rid_to_slide[rid]) + for rid in referenced_rids + if rid in rid_to_slide + } + + +class RefusedToClean(Exception): + """The package does not look the way a readable package should.""" + + +def remove_orphaned_slides(unpacked_dir: Path) -> list[str]: + slides_dir = unpacked_dir / "ppt" / "slides" + slides_rels_dir = slides_dir / "_rels" + pres_rels_path = unpacked_dir / "ppt" / "_rels" / "presentation.xml.rels" + + if not slides_dir.exists(): + return [] + + referenced_slides = get_slides_in_sldidlst(unpacked_dir) + on_disk = sorted(slides_dir.glob("slide*.xml")) + + if on_disk and not any(s.name in referenced_slides for s in on_disk): + listed = re.findall( + r']*r:id="([^"]+)"', + (unpacked_dir / "ppt" / "presentation.xml").read_text(encoding="utf-8") + if (unpacked_dir / "ppt" / "presentation.xml").exists() + else "", + ) + if listed: + raise RefusedToClean( + f" lists {len(listed)} slide(s) and none of the " + f"{len(on_disk)} slide(s) on disk match any of them. Refusing to " + f"delete them all — this is a parse failure, not an empty deck." + ) + + removed = [] + + for slide_file in on_disk: + if slide_file.name not in referenced_slides: + rel_path = slide_file.relative_to(unpacked_dir) + slide_file.unlink() + removed.append(str(rel_path)) + + rels_file = slides_rels_dir / f"{slide_file.name}.rels" + if rels_file.exists(): + rels_file.unlink() + removed.append(str(rels_file.relative_to(unpacked_dir))) + + if removed and pres_rels_path.exists(): + rels_dom = defusedxml.minidom.parse(str(pres_rels_path)) + source_part = rels_source_part(pres_rels_path, unpacked_dir) + changed = False + + for rel in list(rels_dom.getElementsByTagName("Relationship")): + if rel.getAttribute("Type") != SLIDE_REL_TYPE: + continue + part = opc_target( + rel.getAttribute("Target"), source_part, rel.getAttribute("TargetMode") + ) + if part is None: + continue + if posixpath.basename(part) not in referenced_slides: + if rel.parentNode: + rel.parentNode.removeChild(rel) + changed = True + + if changed: + with open(pres_rels_path, "wb") as f: + f.write(rels_dom.toxml(encoding="utf-8")) + + return removed + + +def remove_trash_directory(unpacked_dir: Path) -> list[str]: + trash_dir = unpacked_dir / "[trash]" + removed = [] + + if trash_dir.exists() and trash_dir.is_dir(): + for file_path in trash_dir.iterdir(): + if file_path.is_file(): + rel_path = file_path.relative_to(unpacked_dir) + removed.append(str(rel_path)) + file_path.unlink() + trash_dir.rmdir() + + return removed + + +def _referenced_by(rels_files, unpacked_dir: Path) -> set: + referenced = set() + + for rels_file in rels_files: + source_part = rels_source_part(rels_file, unpacked_dir) + dom = defusedxml.minidom.parse(str(rels_file)) + for rel in dom.getElementsByTagName("Relationship"): + part = opc_target( + rel.getAttribute("Target"), source_part, rel.getAttribute("TargetMode") + ) + if part is not None: + referenced.add(Path(part)) + + return referenced + + +def remove_orphaned_rels_files(unpacked_dir: Path) -> list[str]: + resource_dirs = ["charts", "diagrams", "drawings"] + removed = [] + + for dir_name in resource_dirs: + rels_dir = unpacked_dir / "ppt" / dir_name / "_rels" + if not rels_dir.exists(): + continue + + for rels_file in rels_dir.glob("*.rels"): + resource_file = rels_dir.parent / rels_file.name.replace(".rels", "") + if not resource_file.exists(): + rels_file.unlink() + removed.append(str(rels_file.relative_to(unpacked_dir))) + + return removed + + +def get_referenced_files(unpacked_dir: Path) -> set: + return _referenced_by(sorted(unpacked_dir.rglob("*.rels")), unpacked_dir) + + +def remove_orphaned_files(unpacked_dir: Path, referenced: set) -> list[str]: + resource_dirs = ["media", "embeddings", "charts", "diagrams", "tags", "drawings", "ink"] + removed = [] + + for dir_name in resource_dirs: + dir_path = unpacked_dir / "ppt" / dir_name + if not dir_path.exists(): + continue + + for file_path in dir_path.glob("*"): + if not file_path.is_file(): + continue + rel_path = file_path.relative_to(unpacked_dir) + if rel_path not in referenced: + file_path.unlink() + removed.append(str(rel_path)) + + theme_dir = unpacked_dir / "ppt" / "theme" + if theme_dir.exists(): + for file_path in theme_dir.glob("theme*.xml"): + rel_path = file_path.relative_to(unpacked_dir) + if rel_path not in referenced: + file_path.unlink() + removed.append(str(rel_path)) + theme_rels = theme_dir / "_rels" / f"{file_path.name}.rels" + if theme_rels.exists(): + theme_rels.unlink() + removed.append(str(theme_rels.relative_to(unpacked_dir))) + + notes_dir = unpacked_dir / "ppt" / "notesSlides" + if notes_dir.exists(): + for file_path in notes_dir.glob("*.xml"): + if not file_path.is_file(): + continue + rel_path = file_path.relative_to(unpacked_dir) + if rel_path not in referenced: + file_path.unlink() + removed.append(str(rel_path)) + + notes_rels_dir = notes_dir / "_rels" + if notes_rels_dir.exists(): + for file_path in notes_rels_dir.glob("*.rels"): + notes_file = notes_dir / file_path.name.replace(".rels", "") + if not notes_file.exists(): + file_path.unlink() + removed.append(str(file_path.relative_to(unpacked_dir))) + + return removed + + +def update_content_types(unpacked_dir: Path, removed_files: list[str]) -> None: + ct_path = unpacked_dir / "[Content_Types].xml" + if not ct_path.exists(): + return + + dom = defusedxml.minidom.parse(str(ct_path)) + changed = False + + for override in list(dom.getElementsByTagName("Override")): + part_name = override.getAttribute("PartName").lstrip("/") + if part_name in removed_files: + if override.parentNode: + override.parentNode.removeChild(override) + changed = True + + if changed: + with open(ct_path, "wb") as f: + f.write(dom.toxml(encoding="utf-8")) + + +def clean_unused_files(unpacked_dir: Path) -> list[str]: + all_removed = [] + + if list(unpacked_dir.rglob("*.rels")) and not get_referenced_files(unpacked_dir): + raise RefusedToClean( + "no relationship in this package names a part we can resolve. " + "Refusing to treat every file as unreferenced." + ) + + slides_removed = remove_orphaned_slides(unpacked_dir) + all_removed.extend(slides_removed) + + trash_removed = remove_trash_directory(unpacked_dir) + all_removed.extend(trash_removed) + + while True: + removed_rels = remove_orphaned_rels_files(unpacked_dir) + referenced = get_referenced_files(unpacked_dir) + removed_files = remove_orphaned_files(unpacked_dir, referenced) + + total_removed = removed_rels + removed_files + if not total_removed: + break + + all_removed.extend(total_removed) + + if all_removed: + update_content_types(unpacked_dir, all_removed) + + return all_removed + + +if __name__ == "__main__": + if len(sys.argv) != 2: + print("Usage: python clean.py ", file=sys.stderr) + print("Example: python clean.py unpacked/", file=sys.stderr) + sys.exit(1) + + unpacked_dir = Path(sys.argv[1]) + + if not unpacked_dir.exists(): + print(f"Error: {unpacked_dir} not found", file=sys.stderr) + sys.exit(1) + + try: + removed = clean_unused_files(unpacked_dir) + except (RefusedToClean, ValueError) as e: + print(f"Error: {e}", file=sys.stderr) + print("Nothing was deleted.", file=sys.stderr) + sys.exit(1) + + if removed: + print(f"Removed {len(removed)} unreferenced files:") + for f in removed: + print(f" {f}") + else: + print("No unreferenced files found") diff --git a/.github/skills/anthropic-pptx/scripts/office/helpers/__init__.py b/.github/skills/anthropic-pptx/scripts/office/helpers/__init__.py new file mode 100644 index 00000000..d3c5817c --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/helpers/__init__.py @@ -0,0 +1,111 @@ +import os +import posixpath +import re +import stat +import tempfile +import urllib.parse +import zipfile +from pathlib import Path + +OOXML_FAMILY = { + ".docx": "docx", + ".dotx": "docx", + ".pptx": "pptx", + ".potx": "pptx", + ".xlsx": "xlsx", + ".xltx": "xlsx", +} + +_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.\-]*:") + +SLIDE_REL_TYPE = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/slide" + + +def opc_target(target: str, source_part: str, target_mode: str = "") -> str | None: + if not target: + return None + if target_mode.lower() == "external": + return None + if _SCHEME_RE.match(target): + return None + + target = urllib.parse.unquote(target) + + if "\\" in target: + raise ValueError(f"relationship target is not a POSIX part name: {target!r}") + + if target.startswith("/"): + joined = target.lstrip("/") + else: + joined = posixpath.join(posixpath.dirname(source_part), target) + + parts: list[str] = [] + for segment in posixpath.normpath(joined).split("/"): + if segment in ("", "."): + continue + if segment == "..": + if not parts: + raise ValueError(f"relationship target escapes the package: {target!r}") + parts.pop() + else: + parts.append(segment) + + if not parts: + raise ValueError(f"relationship target resolves to nothing: {target!r}") + return "/".join(parts) + + +def rels_source_part(rels_file: Path, unpacked_dir: Path) -> str: + owner_dir = rels_file.parent.parent.relative_to(unpacked_dir) + return posixpath.join(owner_dir.as_posix(), rels_file.name[: -len(".rels")]).lstrip("./") + + +def part_text(data: bytes) -> str: + return data.decode("utf-8", "surrogateescape") + + +XML_SPACE = " \t\r\n" + + +def rendered_text(text: str, preserve: bool) -> str: + return text if preserve else text.strip(XML_SPACE) + + +def safe_extract(zf: zipfile.ZipFile, dest: Path) -> None: + dest = dest.resolve() + for m in zf.infolist(): + if stat.S_ISLNK(m.external_attr >> 16): + raise ValueError(f"symlink archive entry not allowed: {m.filename!r}") + target = (dest / m.filename).resolve() + if not target.is_relative_to(dest): + raise ValueError(f"unsafe archive entry: {m.filename!r}") + zf.extract(m, dest) + + +def rezip(src_dir: Path, out_path: Path) -> None: + files = sorted(p for p in src_dir.rglob("*") if p.is_file()) + ct = src_dir / "[Content_Types].xml" + fd, tmp_name = tempfile.mkstemp( + prefix=out_path.name + ".", suffix=".tmp", dir=out_path.parent + ) + tmp_out = Path(tmp_name) + try: + with os.fdopen(fd, "wb") as fh: + with zipfile.ZipFile(fh, "w", zipfile.ZIP_DEFLATED) as zf: + if ct.exists(): + zf.write(ct, ct.relative_to(src_dir), compress_type=zipfile.ZIP_STORED) + for f in files: + if f == ct: + continue + zf.write(f, f.relative_to(src_dir)) + if out_path.exists(): + mode = out_path.stat().st_mode & 0o777 + else: + umask = os.umask(0) + os.umask(umask) + mode = 0o666 & ~umask + os.chmod(tmp_out, mode) + os.replace(tmp_out, out_path) + finally: + if tmp_out.exists(): + tmp_out.unlink() diff --git a/.github/skills/anthropic-pptx/scripts/office/helpers/pptx_chart.py b/.github/skills/anthropic-pptx/scripts/office/helpers/pptx_chart.py new file mode 100644 index 00000000..209cb7c5 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/helpers/pptx_chart.py @@ -0,0 +1,170 @@ +"""Find chart XML that PowerPoint refuses but the schema accepts. + +Detection only: for either fault more than one repair is valid, and only the +author knows which was meant. +""" + + +from __future__ import annotations + +import re +from typing import Mapping + +from . import part_text + + +_CHART_PART_RE = re.compile(r"ppt/charts/chart\d+\.xml") + +_GROUPING_RE = re.compile(r"""]*?\bval=["'](\w+)["']""") +_DLBL_POS_RE = re.compile(r"""]*?\bval=["'](\w+)["']""") + +def _strip_ext_lst(text: str) -> str: + out, cursor = [], 0 + for lo, hi in _ext_lst_spans(text): + out.append(text[cursor:lo]) + cursor = hi + out.append(text[cursor:]) + return "".join(out) + +_BAR_GROUP_RE = re.compile(r"]*(?.*?", re.DOTALL) + +STACKED_GROUPINGS = frozenset({"stacked", "percentStacked"}) +ILLEGAL_ON_STACKED = frozenset({"outEnd"}) +LEGAL_ON_STACKED = ("ctr", "inEnd", "inBase") + + +def _check_stacked_label_positions(part: str, xml: str) -> list[str]: + problems: list[str] = [] + for match in _BAR_GROUP_RE.finditer(xml): + block = _strip_ext_lst(match.group(0)) + group = match.group(1) + + grouping = _GROUPING_RE.search(block) + if grouping is None or grouping.group(1) not in STACKED_GROUPINGS: + continue + + bad = [p for p in _DLBL_POS_RE.findall(block) if p in ILLEGAL_ON_STACKED] + for pos in sorted(set(bad)): + problems.append( + f'{part}: {bad.count(pos)} data label(s) use dLblPos="{pos}" on a ' + f"{grouping.group(1)} {group}; PowerPoint allows only " + f"{', '.join(LEGAL_ON_STACKED)} there" + ) + return problems + + + +_ANY_CHART_GROUP_RE = re.compile(r"]*(?.*?", re.DOTALL) + +_AXID_RE = re.compile( + r"""\s*]*?\bval=["'](-?\d+)["']\s*(?:/>|>\s*)""" +) + +_AXIS_DECL_RE = re.compile( + r"""]*(?\s*]*?\bval=["'](-?\d+)["']""" +) + +AXID_LIMIT = { + "barChart": 2, "lineChart": 2, "areaChart": 2, "scatterChart": 2, + "bubbleChart": 2, "radarChart": 2, "stockChart": 2, + "bar3DChart": 3, "line3DChart": 3, "area3DChart": 3, + "surfaceChart": 3, "surface3DChart": 3, +} + +AXID_MINIMUM = { + "barChart": 2, "lineChart": 2, "areaChart": 2, "scatterChart": 2, + "bubbleChart": 2, "radarChart": 2, "stockChart": 2, + "bar3DChart": 2, "area3DChart": 2, "surfaceChart": 2, + "line3DChart": 3, "surface3DChart": 3, +} + + +def _declared_axes(xml: str) -> dict[str, list[str]]: + axes: dict[str, list[str]] = {} + for kind, axid in _AXIS_DECL_RE.findall(xml): + axes.setdefault(kind, []).append(axid) + return axes + + +def _canonical_ids(axes: dict[str, list[str]], limit: int) -> list[str] | None: + category = axes.get("catAx", []) + axes.get("dateAx", []) + value = axes.get("valAx", []) + series = axes.get("serAx", []) + if len(category) != 1 or len(value) != 1 or len(series) > 1: + return None + ids = [category[0], value[0]] + if limit >= 3 and series: + ids.append(series[0]) + return ids + + +def _undeclared_axes(kind: str, block: str, axes: dict[str, list[str]]) -> list[str] | None: + if kind not in AXID_LIMIT: + return None + ids = _AXID_RE.findall(block) + declared = {i for group in axes.values() for i in group} + if len([i for i in ids if i in declared]) >= 2: + return None + return ids + + +def _check_chart_axis_references(part: str, xml: str) -> list[str]: + axes = _declared_axes(xml) + problems: list[str] = [] + declared = {i for group in axes.values() for i in group} + for match in _ANY_CHART_GROUP_RE.finditer(xml): + kind, block = match.group(1), match.group(0) + ids = _undeclared_axes(kind, block, axes) + if ids is None: + continue + if not ids: + problems.append( + f"{part}: declares no this part can resolve; a chart " + f"group needs {AXID_MINIMUM[kind]}, and PowerPoint discards one with fewer" + ) + continue + dead = [i for i in ids if i not in declared] + canonical = _canonical_ids(axes, AXID_LIMIT[kind]) + if canonical is not None and len(canonical) >= AXID_MINIMUM[kind]: + hint = f"Fix: point them at the axes this part declares ({', '.join(canonical)})" + else: + hint = ("Fix: the part declares several axes of a kind -- declare the " + "secondary axes the series expects, or drop them") + detail = (f"of which {', '.join(dead)} name no declared axis" + if dead else f"only {len(ids)} of which this part declares") + problems.append( + f"{part}: references axId {', '.join(ids)}, {detail}, " + f"leaving fewer than two live axes; PowerPoint discards the chart. {hint}" + ) + return problems + + +def _ext_lst_spans(text: str) -> list[tuple[int, int]]: + spans: list[tuple[int, int]] = [] + depth = 0 + start = 0 + for match in re.finditer(r"<(/?)c:extLst\b[^>]*?(/?)>", text): + closing, self_closing = match.group(1), match.group(2) + if self_closing: + continue + if closing: + depth -= 1 + if depth == 0: + spans.append((start, match.end())) + else: + if depth == 0: + start = match.start() + depth += 1 + return spans + + +CHART_CHECKS = (_check_stacked_label_positions, _check_chart_axis_references) + + +def find_chart_problems(files: Mapping[str, bytes]) -> list[str]: + problems: list[str] = [] + for part in sorted(n for n in files if _CHART_PART_RE.fullmatch(n)): + xml = part_text(files[part]) + for check in CHART_CHECKS: + problems.extend(check(part, xml)) + return problems diff --git a/.github/skills/anthropic-pptx/scripts/office/helpers/pptx_slide.py b/.github/skills/anthropic-pptx/scripts/office/helpers/pptx_slide.py new file mode 100644 index 00000000..22f9aee0 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/helpers/pptx_slide.py @@ -0,0 +1,60 @@ +"""Pick the slide-XML schema errors PowerPoint refuses the file over. + +A denylist over lxml's messages, so an unrecognised error class is a miss rather +than a false alarm. +""" + + +from __future__ import annotations + +import re + +SLIDE_PART_RE = re.compile( + r"ppt/(slides|slideLayouts|slideMasters|notesSlides|notesMasters|handoutMasters)" + r"/[^/]+\.xml" +) + +FATAL_SLIDE_ERRORS: tuple[tuple[re.Pattern[str], str], ...] = ( + ( + re.compile(r"\}tableStyleId': This element is not expected"), + "two in one (the schema allows one)", + ), + ( + re.compile(r"\}srgbClr', attribute 'val'"), + "a colour that is not six hex digits", + ), + ( + re.compile(r"\}txBody': Missing child element"), + "a with no children", + ), + ( + re.compile(r"\}miter', attribute 'lim'"), + 'a line join with lim="NaN"', + ), + ( + re.compile(r"\}uLnTx': This element is not expected"), + " in a position the schema forbids", + ), + ( + re.compile(r"\}overrideClrMapping': This element is not expected"), + " in a position the schema forbids", + ), + ( + re.compile(r"\}nvGrpSpPr': Missing child element"), + "a with no children", + ), +) + + +def is_schema_verdict(error: str) -> bool: + return error.startswith("Element ") + + +def fatal_slide_errors(errors: set[str]) -> list[str]: + out = [] + for error in sorted(errors): + for pattern, meaning in FATAL_SLIDE_ERRORS: + if pattern.search(error): + out.append(f"{meaning}: {error}") + break + return out diff --git a/.github/skills/anthropic-pptx/scripts/office/helpers/pptx_theme.py b/.github/skills/anthropic-pptx/scripts/office/helpers/pptx_theme.py new file mode 100644 index 00000000..84466201 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/helpers/pptx_theme.py @@ -0,0 +1,114 @@ +"""Find masters sharing a theme part in the way PowerPoint refuses to open. + +Reports only; the fix is to move back to directly after + in ppt/presentation.xml. +""" + + +from __future__ import annotations + +import posixpath +import re +from typing import Mapping + +from . import part_text + +THEME_REL_TYPE = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/theme" + +_MASTER_RE = re.compile( + r"^ppt/(?PslideMasters|notesMasters|handoutMasters)/" + r"(?:slide|notes|handout)Master(?P\d+)\.xml$" +) +_GROUP_ORDER = {"slideMasters": 0, "notesMasters": 1, "handoutMasters": 2} + +_RELATIONSHIP_RE = re.compile( + r"]*?(?:/>|>.*?)", re.DOTALL +) + + +def _sort_key(name: str) -> tuple[int, int]: + m = _MASTER_RE.match(name) + assert m is not None + return (_GROUP_ORDER[m.group("group")], int(m.group("num"))) + + +def _rels_path(part: str) -> str: + directory, base = posixpath.split(part) + return f"{directory}/_rels/{base}.rels" + + +def _resolve(rels_path: str, target: str) -> str: + if target.startswith("/"): + return target.lstrip("/") + part_dir = posixpath.dirname(posixpath.dirname(rels_path)) + return posixpath.normpath(posixpath.join(part_dir, target)) + + +def _theme_rel(files: Mapping[str, bytes], master: str): + rels_path = _rels_path(master) + rels = files.get(rels_path) + if rels is None: + return None + for element in _RELATIONSHIP_RE.findall(part_text(rels)): + if f'Type="{THEME_REL_TYPE}"' not in element: + continue + target = re.search(r'\bTarget="([^"]+)"', element) + if target is None: + continue + return rels_path, element, _resolve(rels_path, target.group(1)) + return None + + +def _masters(files: Mapping[str, bytes]) -> list[str]: + return sorted((n for n in files if _MASTER_RE.match(n)), key=_sort_key) + + +_PRESENTATION = "ppt/presentation.xml" +_NOTES_MASTERS = "ppt/notesMasters/" +_IGNORABLE_RE = re.compile(r"|<\?.*?\?>", re.DOTALL) +_AFTER_SLDIDLST_RE = re.compile( + r"]*/>|[^>]*>.*?)\s*(<[^>\s/]+)", re.DOTALL +) + + +def _notes_master_share_is_inert(files: Mapping[str, bytes]) -> bool: + data = files.get(_PRESENTATION) + if data is None: + return False + match = _AFTER_SLDIDLST_RE.search(_IGNORABLE_RE.sub("", part_text(data))) + return match is not None and match.group(1) == " bool: + return inert_notes and master.startswith(_NOTES_MASTERS) + + +def find_shared_master_themes(files: Mapping[str, bytes]) -> list[str]: + return [ + f"{master} shares {theme} with {first}" + for master, _, _, theme, first in _shares(files) + ] + + +def live_shared_master_themes(files: Mapping[str, bytes]) -> list[str]: + inert_notes = _notes_master_share_is_inert(files) + return [ + f"{master} shares {theme} with {first}" + for master, _, _, theme, first in _shares(files) + if not _is_inert(master, inert_notes) + ] diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chart.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chart.xsd new file mode 100644 index 00000000..6454ef9a --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chart.xsd @@ -0,0 +1,1499 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chartDrawing.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chartDrawing.xsd new file mode 100644 index 00000000..afa4f463 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chartDrawing.xsd @@ -0,0 +1,146 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-diagram.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-diagram.xsd new file mode 100644 index 00000000..64e66b8a --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-diagram.xsd @@ -0,0 +1,1085 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-lockedCanvas.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-lockedCanvas.xsd new file mode 100644 index 00000000..687eea82 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-lockedCanvas.xsd @@ -0,0 +1,11 @@ + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-main.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-main.xsd new file mode 100644 index 00000000..6ac81b06 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-main.xsd @@ -0,0 +1,3081 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-picture.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-picture.xsd new file mode 100644 index 00000000..1dbf0514 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-picture.xsd @@ -0,0 +1,23 @@ + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-spreadsheetDrawing.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-spreadsheetDrawing.xsd new file mode 100644 index 00000000..f1af17db --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-spreadsheetDrawing.xsd @@ -0,0 +1,185 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-wordprocessingDrawing.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-wordprocessingDrawing.xsd new file mode 100644 index 00000000..0a185ab6 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-wordprocessingDrawing.xsd @@ -0,0 +1,287 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/pml.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/pml.xsd new file mode 100644 index 00000000..14ef4888 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/pml.xsd @@ -0,0 +1,1676 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-additionalCharacteristics.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-additionalCharacteristics.xsd new file mode 100644 index 00000000..c20f3bf1 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-additionalCharacteristics.xsd @@ -0,0 +1,28 @@ + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-bibliography.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-bibliography.xsd new file mode 100644 index 00000000..ac602522 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-bibliography.xsd @@ -0,0 +1,144 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-commonSimpleTypes.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-commonSimpleTypes.xsd new file mode 100644 index 00000000..424b8ba8 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-commonSimpleTypes.xsd @@ -0,0 +1,174 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlDataProperties.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlDataProperties.xsd new file mode 100644 index 00000000..2bddce29 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlDataProperties.xsd @@ -0,0 +1,25 @@ + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlSchemaProperties.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlSchemaProperties.xsd new file mode 100644 index 00000000..8a8c18ba --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlSchemaProperties.xsd @@ -0,0 +1,18 @@ + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd new file mode 100644 index 00000000..5c42706a --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd @@ -0,0 +1,59 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd new file mode 100644 index 00000000..853c341c --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd @@ -0,0 +1,56 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesVariantTypes.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesVariantTypes.xsd new file mode 100644 index 00000000..da835ee8 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesVariantTypes.xsd @@ -0,0 +1,195 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-math.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-math.xsd new file mode 100644 index 00000000..87ad2658 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-math.xsd @@ -0,0 +1,582 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-relationshipReference.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-relationshipReference.xsd new file mode 100644 index 00000000..9e86f1b2 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-relationshipReference.xsd @@ -0,0 +1,25 @@ + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/sml.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/sml.xsd new file mode 100644 index 00000000..d0be42e7 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/sml.xsd @@ -0,0 +1,4439 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-main.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-main.xsd new file mode 100644 index 00000000..8821dd18 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-main.xsd @@ -0,0 +1,570 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-officeDrawing.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-officeDrawing.xsd new file mode 100644 index 00000000..ca2575c7 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-officeDrawing.xsd @@ -0,0 +1,509 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-presentationDrawing.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-presentationDrawing.xsd new file mode 100644 index 00000000..dd079e60 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-presentationDrawing.xsd @@ -0,0 +1,12 @@ + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-spreadsheetDrawing.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-spreadsheetDrawing.xsd new file mode 100644 index 00000000..3dd6cf62 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-spreadsheetDrawing.xsd @@ -0,0 +1,108 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-wordprocessingDrawing.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-wordprocessingDrawing.xsd new file mode 100644 index 00000000..f1041e34 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-wordprocessingDrawing.xsd @@ -0,0 +1,96 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/wml.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/wml.xsd new file mode 100644 index 00000000..9c5b7a63 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/wml.xsd @@ -0,0 +1,3646 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/xml.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/xml.xsd new file mode 100644 index 00000000..0f13678d --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ISO-IEC29500-4_2016/xml.xsd @@ -0,0 +1,116 @@ + + + + + + See http://www.w3.org/XML/1998/namespace.html and + http://www.w3.org/TR/REC-xml for information about this namespace. + + This schema document describes the XML namespace, in a form + suitable for import by other schema documents. + + Note that local names in this namespace are intended to be defined + only by the World Wide Web Consortium or its subgroups. The + following names are currently defined in this namespace and should + not be used with conflicting semantics by any Working Group, + specification, or document instance: + + base (as an attribute name): denotes an attribute whose value + provides a URI to be used as the base for interpreting any + relative URIs in the scope of the element on which it + appears; its value is inherited. This name is reserved + by virtue of its definition in the XML Base specification. + + lang (as an attribute name): denotes an attribute whose value + is a language code for the natural language of the content of + any element; its value is inherited. This name is reserved + by virtue of its definition in the XML specification. + + space (as an attribute name): denotes an attribute whose + value is a keyword indicating what whitespace processing + discipline is intended for the content of the element; its + value is inherited. This name is reserved by virtue of its + definition in the XML specification. + + Father (in any context at all): denotes Jon Bosak, the chair of + the original XML Working Group. This name is reserved by + the following decision of the W3C XML Plenary and + XML Coordination groups: + + In appreciation for his vision, leadership and dedication + the W3C XML Plenary on this 10th day of February, 2000 + reserves for Jon Bosak in perpetuity the XML name + xml:Father + + + + + This schema defines attributes and an attribute group + suitable for use by + schemas wishing to allow xml:base, xml:lang or xml:space attributes + on elements they define. + + To enable this, such a schema must import this schema + for the XML namespace, e.g. as follows: + <schema . . .> + . . . + <import namespace="http://www.w3.org/XML/1998/namespace" + schemaLocation="http://www.w3.org/2001/03/xml.xsd"/> + + Subsequently, qualified reference to any of the attributes + or the group defined below will have the desired effect, e.g. + + <type . . .> + . . . + <attributeGroup ref="xml:specialAttrs"/> + + will define a type which will schema-validate an instance + element with any of those attributes + + + + In keeping with the XML Schema WG's standard versioning + policy, this schema document will persist at + http://www.w3.org/2001/03/xml.xsd. + At the date of issue it can also be found at + http://www.w3.org/2001/xml.xsd. + The schema document at that URI may however change in the future, + in order to remain compatible with the latest version of XML Schema + itself. In other words, if the XML Schema namespace changes, the version + of this document at + http://www.w3.org/2001/xml.xsd will change + accordingly; the version at + http://www.w3.org/2001/03/xml.xsd will not change. + + + + + + In due course, we should install the relevant ISO 2- and 3-letter + codes as the enumerated possible values . . . + + + + + + + + + + + + + + + See http://www.w3.org/TR/xmlbase/ for + information about this attribute. + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-contentTypes.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-contentTypes.xsd new file mode 100644 index 00000000..a6de9d27 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-contentTypes.xsd @@ -0,0 +1,42 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-coreProperties.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-coreProperties.xsd new file mode 100644 index 00000000..10e978b6 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-coreProperties.xsd @@ -0,0 +1,50 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-digSig.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-digSig.xsd new file mode 100644 index 00000000..4248bf7a --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-digSig.xsd @@ -0,0 +1,49 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-relationships.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-relationships.xsd new file mode 100644 index 00000000..56497467 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/ecma/fouth-edition/opc-relationships.xsd @@ -0,0 +1,33 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/mce/mc.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/mce/mc.xsd new file mode 100644 index 00000000..ef725457 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/mce/mc.xsd @@ -0,0 +1,75 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-2010.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-2010.xsd new file mode 100644 index 00000000..f65f7777 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-2010.xsd @@ -0,0 +1,560 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-2012.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-2012.xsd new file mode 100644 index 00000000..6b00755a --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-2012.xsd @@ -0,0 +1,67 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-2018.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-2018.xsd new file mode 100644 index 00000000..f321d333 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-2018.xsd @@ -0,0 +1,14 @@ + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-cex-2018.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-cex-2018.xsd new file mode 100644 index 00000000..364c6a9b --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-cex-2018.xsd @@ -0,0 +1,20 @@ + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-cid-2016.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-cid-2016.xsd new file mode 100644 index 00000000..fed9d15b --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-cid-2016.xsd @@ -0,0 +1,13 @@ + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-sdtdatahash-2020.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-sdtdatahash-2020.xsd new file mode 100644 index 00000000..680cf154 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-sdtdatahash-2020.xsd @@ -0,0 +1,4 @@ + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-symex-2015.xsd b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-symex-2015.xsd new file mode 100644 index 00000000..89ada908 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/schemas/microsoft/wml-symex-2015.xsd @@ -0,0 +1,8 @@ + + + + + + + + diff --git a/.github/skills/anthropic-pptx/scripts/office/soffice.py b/.github/skills/anthropic-pptx/scripts/office/soffice.py new file mode 100644 index 00000000..0b4c99de --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/soffice.py @@ -0,0 +1,192 @@ +""" +Helper for running LibreOffice (soffice) in environments where AF_UNIX +sockets may be blocked (e.g., sandboxed VMs). Detects the restriction +at runtime and applies an LD_PRELOAD shim if needed. + +Usage: + from office.soffice import run_soffice + + result = run_soffice(["--headless", "--convert-to", "pdf", "input.docx"]) + +Call soffice through run_soffice, not through subprocess with get_soffice_env(): +the env dict carries the shim but names no user profile, and a non-root sandbox +cannot bootstrap the default one -- soffice aborts with "User installation could +not be completed" and converts nothing. get_soffice_env() stays public for the +callers that build their own argv (they must pass -env:UserInstallation too). +""" + +import contextlib +import os +import socket +import subprocess +import tempfile +from collections.abc import Iterable +from pathlib import Path + + +def get_soffice_env() -> dict: + env = os.environ.copy() + env["SAL_USE_VCLPLUGIN"] = "svp" + + if _needs_shim(): + shim = _ensure_shim() + env["LD_PRELOAD"] = str(shim) + + return env + + +def run_soffice(args: Iterable[str], **kwargs) -> subprocess.CompletedProcess: + args = list(args) + with contextlib.ExitStack() as stack: + if not any(str(a).startswith("-env:UserInstallation") for a in args): + profile = stack.enter_context( + tempfile.TemporaryDirectory(prefix="lo_profile_", ignore_cleanup_errors=True) + ) + args = [f"-env:UserInstallation={Path(profile).as_uri()}"] + args + return subprocess.run(["soffice"] + args, env=get_soffice_env(), **kwargs) + + + +_SHIM_SO = Path(tempfile.gettempdir()) / "lo_socket_shim.so" + + +def _needs_shim() -> bool: + try: + s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + s.close() + return False + except OSError: + return True + + +def _ensure_shim() -> Path: + if _SHIM_SO.exists(): + return _SHIM_SO + + src = Path(tempfile.gettempdir()) / "lo_socket_shim.c" + src.write_text(_SHIM_SOURCE) + subprocess.run( + ["gcc", "-shared", "-fPIC", "-o", str(_SHIM_SO), str(src), "-ldl"], + check=True, + capture_output=True, + ) + src.unlink() + return _SHIM_SO + + + +_SHIM_SOURCE = r""" +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include + +static int (*real_socket)(int, int, int); +static int (*real_socketpair)(int, int, int, int[2]); +static int (*real_listen)(int, int); +static int (*real_accept)(int, struct sockaddr *, socklen_t *); +static int (*real_close)(int); +static int (*real_read)(int, void *, size_t); + +/* Per-FD bookkeeping (FDs >= 1024 are passed through unshimmed). */ +static int is_shimmed[1024]; +static int peer_of[1024]; +static int wake_r[1024]; /* accept() blocks reading this */ +static int wake_w[1024]; /* close() writes to this */ +static int listener_fd = -1; /* FD that received listen() */ + +__attribute__((constructor)) +static void init(void) { + real_socket = dlsym(RTLD_NEXT, "socket"); + real_socketpair = dlsym(RTLD_NEXT, "socketpair"); + real_listen = dlsym(RTLD_NEXT, "listen"); + real_accept = dlsym(RTLD_NEXT, "accept"); + real_close = dlsym(RTLD_NEXT, "close"); + real_read = dlsym(RTLD_NEXT, "read"); + for (int i = 0; i < 1024; i++) { + peer_of[i] = -1; + wake_r[i] = -1; + wake_w[i] = -1; + } +} + +/* ---- socket ---------------------------------------------------------- */ +int socket(int domain, int type, int protocol) { + if (domain == AF_UNIX) { + int fd = real_socket(domain, type, protocol); + if (fd >= 0) return fd; + /* socket(AF_UNIX) blocked – fall back to socketpair(). */ + int sv[2]; + if (real_socketpair(domain, type, protocol, sv) == 0) { + if (sv[0] >= 0 && sv[0] < 1024) { + is_shimmed[sv[0]] = 1; + peer_of[sv[0]] = sv[1]; + int wp[2]; + if (pipe(wp) == 0) { + wake_r[sv[0]] = wp[0]; + wake_w[sv[0]] = wp[1]; + } + } + return sv[0]; + } + errno = EPERM; + return -1; + } + return real_socket(domain, type, protocol); +} + +/* ---- listen ---------------------------------------------------------- */ +int listen(int sockfd, int backlog) { + if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) { + listener_fd = sockfd; + return 0; + } + return real_listen(sockfd, backlog); +} + +/* ---- accept ---------------------------------------------------------- */ +int accept(int sockfd, struct sockaddr *addr, socklen_t *addrlen) { + if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) { + /* Block until close() writes to the wake pipe. */ + if (wake_r[sockfd] >= 0) { + char buf; + real_read(wake_r[sockfd], &buf, 1); + } + errno = ECONNABORTED; + return -1; + } + return real_accept(sockfd, addr, addrlen); +} + +/* ---- close ----------------------------------------------------------- */ +int close(int fd) { + if (fd >= 0 && fd < 1024 && is_shimmed[fd]) { + int was_listener = (fd == listener_fd); + is_shimmed[fd] = 0; + + if (wake_w[fd] >= 0) { /* unblock accept() */ + char c = 0; + write(wake_w[fd], &c, 1); + real_close(wake_w[fd]); + wake_w[fd] = -1; + } + if (wake_r[fd] >= 0) { real_close(wake_r[fd]); wake_r[fd] = -1; } + if (peer_of[fd] >= 0) { real_close(peer_of[fd]); peer_of[fd] = -1; } + + if (was_listener) + _exit(0); /* conversion done – exit */ + } + return real_close(fd); +} +""" + + + +if __name__ == "__main__": + import sys + result = run_soffice(sys.argv[1:]) + sys.exit(result.returncode) diff --git a/.github/skills/anthropic-pptx/scripts/office/validate.py b/.github/skills/anthropic-pptx/scripts/office/validate.py new file mode 100644 index 00000000..29ca186a --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/validate.py @@ -0,0 +1,173 @@ +""" +Command line tool to validate Office document XML files against XSD schemas and tracked changes. + +Usage: + python validate.py [--original ] [--auto-repair] [--author NAME] + +The first argument can be either: +- An unpacked directory containing the Office document XML files +- A packed Office file (.docx/.pptx/.xlsx or .dotx/.potx/.xltx template) which will be unpacked to a temp directory + +Auto-repair fixes: +- paraId/durableId values that exceed OOXML limits +- Missing xml:space="preserve" on w:t elements with whitespace +""" + +import argparse +import sys +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.ElementTree as ET +from defusedxml.common import DefusedXmlException + +from helpers import OOXML_FAMILY, rezip, safe_extract +from validators import DOCXSchemaValidator, PPTXSchemaValidator, RedliningValidator + +WORD_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + + +def _fail(message: str): + print(f"Error: {message}", file=sys.stderr) + sys.exit(2) + + +def _has_tracked_changes(unpacked_dir: Path) -> bool: + document = unpacked_dir / "word" / "document.xml" + if not document.is_file(): + return False + try: + root = ET.parse(document).getroot() + except (ET.ParseError, DefusedXmlException): + return False + tracked = {f"{{{WORD_NS}}}ins", f"{{{WORD_NS}}}del"} + return any(elem.tag in tracked for elem in root.iter()) + + +def main(): + parser = argparse.ArgumentParser(description="Validate Office document XML files") + parser.add_argument( + "path", + help="Path to unpacked directory or packed Office file (.docx/.pptx/.xlsx or .dotx/.potx/.xltx)", + ) + parser.add_argument( + "--original", + required=False, + default=None, + help="Path to original file (.docx/.pptx/.xlsx or .dotx/.potx/.xltx). If omitted, all XSD errors are reported and redlining validation is skipped.", + ) + parser.add_argument( + "-v", + "--verbose", + action="store_true", + help="Enable verbose output", + ) + parser.add_argument( + "--auto-repair", + action="store_true", + help="Automatically repair common issues (hex IDs, whitespace preservation). " + "Modifies the input in place: repairs to a packed file are written back to it.", + ) + parser.add_argument( + "--author", + default=None, + help="The name you are redlining under. Passing it turns on the " + "tracked-change check: any text differing from --original without a " + "/ recording it is reported. Untracked edits carry no " + "author, so the check covers them whoever made them — the name marks " + "the run as redlining work and is not used to filter. Requires " + "--original; docx only.", + ) + args = parser.parse_args() + + if args.author is not None and not args.original: + _fail("--author requires --original") + + path = Path(args.path) + if not path.exists(): + _fail(f"{path} does not exist") + + original_file = None + if args.original: + original_file = Path(args.original) + if not original_file.is_file(): + _fail(f"{original_file} is not a file") + if original_file.suffix.lower() not in OOXML_FAMILY: + _fail(f"{original_file} must be one of: {', '.join(sorted(OOXML_FAMILY))}") + + family = OOXML_FAMILY.get((original_file or path).suffix.lower()) + if family is None: + _fail( + f"Cannot determine file type from {path}. Use --original or provide one of: {', '.join(sorted(OOXML_FAMILY))}." + ) + + if args.author is not None and family != "docx": + _fail(f"--author only applies to docx files, not {family}") + + packed_file = None + temp_dir_ctx = None + if path.is_file() and path.suffix.lower() in OOXML_FAMILY: + packed_file = path + temp_dir_ctx = tempfile.TemporaryDirectory() + unpacked_dir = Path(temp_dir_ctx.name) + try: + with zipfile.ZipFile(path, "r") as zf: + safe_extract(zf, unpacked_dir) + except (zipfile.BadZipFile, ValueError, OSError) as e: + _fail(f"cannot unpack {path}: {e}") + else: + if not path.is_dir(): + _fail(f"{path} is not a directory or Office file") + unpacked_dir = path + + match family: + case "docx": + validators = [ + DOCXSchemaValidator(unpacked_dir, original_file, verbose=args.verbose), + ] + if args.author is not None: + validators.append( + RedliningValidator(unpacked_dir, original_file, verbose=args.verbose) + ) + elif original_file and _has_tracked_changes(unpacked_dir): + print( + "Note: this document has tracked changes; they were not " + "checked against the original (pass --author to check)." + ) + case "pptx": + validators = [ + PPTXSchemaValidator(unpacked_dir, original_file, verbose=args.verbose), + ] + case "xlsx": + exts = ", ".join(k for k, v in sorted(OOXML_FAMILY.items()) if v == "xlsx") + print( + f"No XSD schema validation is performed for xlsx-family files ({exts}). " + "For formula-error checking, use scripts/recalc.py instead." + ) + sys.exit(0) + case _: + print(f"Error: Validation not supported for file type {family}") + sys.exit(1) + + if args.auto_repair: + total_repairs = sum(v.repair() for v in validators) + if total_repairs: + print(f"Auto-repaired {total_repairs} issue(s)") + if packed_file is not None: + rezip(unpacked_dir, packed_file) + print(f"Wrote repaired file to {packed_file}") + + success = all([v.validate() for v in validators]) + + if temp_dir_ctx is not None: + temp_dir_ctx.cleanup() + + if success: + print("All validations PASSED!") + + sys.exit(0 if success else 1) + + +if __name__ == "__main__": + main() diff --git a/.github/skills/anthropic-pptx/scripts/office/validators/__init__.py b/.github/skills/anthropic-pptx/scripts/office/validators/__init__.py new file mode 100644 index 00000000..db092ece --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/validators/__init__.py @@ -0,0 +1,15 @@ +""" +Validation modules for Word document processing. +""" + +from .base import BaseSchemaValidator +from .docx import DOCXSchemaValidator +from .pptx import PPTXSchemaValidator +from .redlining import RedliningValidator + +__all__ = [ + "BaseSchemaValidator", + "DOCXSchemaValidator", + "PPTXSchemaValidator", + "RedliningValidator", +] diff --git a/.github/skills/anthropic-pptx/scripts/office/validators/base.py b/.github/skills/anthropic-pptx/scripts/office/validators/base.py new file mode 100644 index 00000000..33fc97bb --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/validators/base.py @@ -0,0 +1,875 @@ +""" +Base validator with common validation logic for document files. +""" + +import re +from pathlib import Path + +import defusedxml.minidom +from functools import lru_cache + +import lxml.etree + +from helpers import safe_extract + + +@lru_cache(maxsize=None) +def _load_schema(schema_path: str): + with open(schema_path, "rb") as xsd_file: + xsd_doc = lxml.etree.parse( + xsd_file, parser=lxml.etree.XMLParser(), base_url=schema_path + ) + return lxml.etree.XMLSchema(xsd_doc) + +class BaseSchemaValidator: + + IGNORED_VALIDATION_ERRORS = [ + "hyphenationZone", + "purl.org/dc/terms", + ] + + UNIQUE_ID_REQUIREMENTS = { + "comment": ("id", "file"), + "commentrangestart": ("id", "file"), + "commentrangeend": ("id", "file"), + "bookmarkstart": ("id", "file"), + "bookmarkend": ("id", "file"), + "sldid": ("id", "file"), + "sldmasterid": ("id", "global"), + "sldlayoutid": ("id", "global"), + "cm": ("authorid", "file"), + "sheet": ("sheetid", "file"), + "definedname": ("id", "file"), + "cxnsp": ("id", "file"), + "sp": ("id", "file"), + "pic": ("id", "file"), + "grpsp": ("id", "file"), + } + + EXCLUDED_ID_CONTAINERS = { + "sectionlst", + } + + ELEMENT_RELATIONSHIP_TYPES = {} + + SCHEMA_MAPPINGS = { + "word": "ISO-IEC29500-4_2016/wml.xsd", + "ppt": "ISO-IEC29500-4_2016/pml.xsd", + "xl": "ISO-IEC29500-4_2016/sml.xsd", + "[Content_Types].xml": "ecma/fouth-edition/opc-contentTypes.xsd", + "app.xml": "ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd", + "core.xml": "ecma/fouth-edition/opc-coreProperties.xsd", + "custom.xml": "ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd", + ".rels": "ecma/fouth-edition/opc-relationships.xsd", + "people.xml": "microsoft/wml-2012.xsd", + "commentsIds.xml": "microsoft/wml-cid-2016.xsd", + "commentsExtensible.xml": "microsoft/wml-cex-2018.xsd", + "commentsExtended.xml": "microsoft/wml-2012.xsd", + "chart": "ISO-IEC29500-4_2016/dml-chart.xsd", + "theme": "ISO-IEC29500-4_2016/dml-main.xsd", + "drawing": "ISO-IEC29500-4_2016/dml-main.xsd", + } + + MC_NAMESPACE = "http://schemas.openxmlformats.org/markup-compatibility/2006" + XML_NAMESPACE = "http://www.w3.org/XML/1998/namespace" + + PACKAGE_RELATIONSHIPS_NAMESPACE = ( + "http://schemas.openxmlformats.org/package/2006/relationships" + ) + OFFICE_RELATIONSHIPS_NAMESPACE = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships" + ) + CONTENT_TYPES_NAMESPACE = ( + "http://schemas.openxmlformats.org/package/2006/content-types" + ) + + MAIN_CONTENT_FOLDERS = {"word", "ppt", "xl"} + + OOXML_NAMESPACES = { + "http://schemas.openxmlformats.org/officeDocument/2006/math", + "http://schemas.openxmlformats.org/officeDocument/2006/relationships", + "http://schemas.openxmlformats.org/schemaLibrary/2006/main", + "http://schemas.openxmlformats.org/drawingml/2006/main", + "http://schemas.openxmlformats.org/drawingml/2006/chart", + "http://schemas.openxmlformats.org/drawingml/2006/chartDrawing", + "http://schemas.openxmlformats.org/drawingml/2006/diagram", + "http://schemas.openxmlformats.org/drawingml/2006/picture", + "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing", + "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing", + "http://schemas.openxmlformats.org/wordprocessingml/2006/main", + "http://schemas.openxmlformats.org/presentationml/2006/main", + "http://schemas.openxmlformats.org/spreadsheetml/2006/main", + "http://schemas.openxmlformats.org/officeDocument/2006/sharedTypes", + "http://www.w3.org/XML/1998/namespace", + } + + def __init__(self, unpacked_dir, original_file=None, verbose=False): + self.unpacked_dir = Path(unpacked_dir).resolve() + self.original_file = Path(original_file) if original_file else None + self.verbose = verbose + + self.schemas_dir = Path(__file__).parent.parent / "schemas" + + patterns = ["*.xml", "*.rels"] + self.xml_files = [ + f for pattern in patterns for f in self.unpacked_dir.rglob(pattern) + ] + + if not self.xml_files: + print(f"Warning: No XML files found in {self.unpacked_dir}") + + def validate(self): + raise NotImplementedError("Subclasses must implement the validate method") + + def repair(self) -> int: + return self.repair_whitespace_preservation() + + def repair_whitespace_preservation(self) -> int: + repairs = 0 + + for xml_file in self.xml_files: + try: + content = xml_file.read_text(encoding="utf-8") + dom = defusedxml.minidom.parseString(content) + pending = [] + + for elem in dom.getElementsByTagName("*"): + local_name = elem.tagName.rsplit(":", 1)[-1] + if local_name in ("t", "delText", "instrText", "delInstrText"): + text = "".join( + child.data + for child in elem.childNodes + if child.nodeType in (child.TEXT_NODE, child.CDATA_SECTION_NODE) + ) + ws = (" ", "\t", "\n", "\r") + if text and (text.startswith(ws) or text.endswith(ws)): + if elem.getAttribute("xml:space") != "preserve": + elem.setAttribute("xml:space", "preserve") + text_preview = repr(text[:30]) + "..." if len(text) > 30 else repr(text) + pending.append(f" Repaired: {xml_file.name}: Added xml:space='preserve' to {elem.tagName}: {text_preview}") + + if pending: + xml_file.write_bytes(dom.toxml(encoding="UTF-8")) + for message in pending: + print(message) + repairs += len(pending) + + except Exception: + pass + + return repairs + + def validate_xml(self): + errors = [] + + for xml_file in self.xml_files: + try: + lxml.etree.parse(str(xml_file)) + except lxml.etree.XMLSyntaxError as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {e.lineno}: {e.msg}" + ) + except Exception as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Unexpected error: {str(e)}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} XML violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All XML files are well-formed") + return True + + def validate_namespaces(self): + errors = [] + + for xml_file in self.xml_files: + try: + root = lxml.etree.parse(str(xml_file)).getroot() + declared = set(root.nsmap.keys()) - {None} + + for attr_val in [ + v for k, v in root.attrib.items() if k.endswith("Ignorable") + ]: + undeclared = set(attr_val.split()) - declared + errors.extend( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Namespace '{ns}' in Ignorable but not declared" + for ns in undeclared + ) + except lxml.etree.XMLSyntaxError: + continue + + if errors: + print(f"FAILED - {len(errors)} namespace issues:") + for error in errors: + print(error) + return False + if self.verbose: + print("PASSED - All namespace prefixes properly declared") + return True + + def validate_unique_ids(self): + errors = [] + global_ids = {} + + for xml_file in self.xml_files: + try: + root = lxml.etree.parse(str(xml_file)).getroot() + file_ids = {} + + mc_elements = root.xpath( + ".//mc:AlternateContent", namespaces={"mc": self.MC_NAMESPACE} + ) + for elem in mc_elements: + elem.getparent().remove(elem) + + for elem in root.iter(): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + tag = ( + elem.tag.split("}")[-1].lower() + if "}" in elem.tag + else elem.tag.lower() + ) + + if tag in self.UNIQUE_ID_REQUIREMENTS: + in_excluded_container = any( + ancestor.tag.split("}")[-1].lower() in self.EXCLUDED_ID_CONTAINERS + for ancestor in elem.iterancestors() + ) + if in_excluded_container: + continue + + attr_name, scope = self.UNIQUE_ID_REQUIREMENTS[tag] + + id_value = None + for attr, value in elem.attrib.items(): + attr_local = ( + attr.split("}")[-1].lower() + if "}" in attr + else attr.lower() + ) + if attr_local == attr_name: + id_value = value + break + + if id_value is not None: + if scope == "global": + if id_value in global_ids: + prev_file, prev_line, prev_tag = global_ids[ + id_value + ] + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: Global ID '{id_value}' in <{tag}> " + f"already used in {prev_file} at line {prev_line} in <{prev_tag}>" + ) + else: + global_ids[id_value] = ( + xml_file.relative_to(self.unpacked_dir), + elem.sourceline, + tag, + ) + elif scope == "file": + key = (tag, attr_name) + if key not in file_ids: + file_ids[key] = {} + + if id_value in file_ids[key]: + prev_line = file_ids[key][id_value] + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: Duplicate {attr_name}='{id_value}' in <{tag}> " + f"(first occurrence at line {prev_line})" + ) + else: + file_ids[key][id_value] = elem.sourceline + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} ID uniqueness violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All required IDs are unique") + return True + + def validate_file_references(self): + errors = [] + + rels_files = list(self.unpacked_dir.rglob("*.rels")) + + if not rels_files: + if self.verbose: + print("PASSED - No .rels files found") + return True + + all_files = [] + for file_path in self.unpacked_dir.rglob("*"): + if ( + file_path.is_file() + and file_path.name != "[Content_Types].xml" + and not file_path.name.endswith(".rels") + ): + all_files.append(file_path.resolve()) + + all_referenced_files = set() + + if self.verbose: + print( + f"Found {len(rels_files)} .rels files and {len(all_files)} target files" + ) + + for rels_file in rels_files: + try: + rels_root = lxml.etree.parse(str(rels_file)).getroot() + + rels_dir = rels_file.parent + + referenced_files = set() + broken_refs = [] + + for rel in rels_root.findall( + ".//ns:Relationship", + namespaces={"ns": self.PACKAGE_RELATIONSHIPS_NAMESPACE}, + ): + target = rel.get("Target") + if rel.get("TargetMode") == "External": + continue + if target and not target.startswith( + ("http", "mailto:") + ): + if target.startswith("/"): + target_path = self.unpacked_dir / target.lstrip("/") + elif rels_file.name == ".rels": + target_path = self.unpacked_dir / target + else: + base_dir = rels_dir.parent + target_path = base_dir / target + + try: + target_path = target_path.resolve() + if target_path.exists() and target_path.is_file(): + referenced_files.add(target_path) + all_referenced_files.add(target_path) + else: + broken_refs.append((target, rel.sourceline)) + except (OSError, ValueError): + broken_refs.append((target, rel.sourceline)) + + if broken_refs: + rel_path = rels_file.relative_to(self.unpacked_dir) + for broken_ref, line_num in broken_refs: + errors.append( + f" {rel_path}: Line {line_num}: Broken reference to {broken_ref}" + ) + + except Exception as e: + rel_path = rels_file.relative_to(self.unpacked_dir) + errors.append(f" Error parsing {rel_path}: {e}") + + unreferenced_files = set(all_files) - all_referenced_files + + if unreferenced_files: + for unref_file in sorted(unreferenced_files): + unref_rel_path = unref_file.relative_to(self.unpacked_dir) + errors.append(f" Unreferenced file: {unref_rel_path}") + + if errors: + print(f"FAILED - Found {len(errors)} relationship validation errors:") + for error in errors: + print(error) + print( + "CRITICAL: These errors will cause the document to appear corrupt. " + + "Broken references MUST be fixed, " + + "and unreferenced files MUST be referenced or removed." + ) + return False + else: + if self.verbose: + print( + "PASSED - All references are valid and all files are properly referenced" + ) + return True + + def validate_all_relationship_ids(self): + import lxml.etree + + errors = [] + + for xml_file in self.xml_files: + if xml_file.suffix == ".rels": + continue + + rels_dir = xml_file.parent / "_rels" + rels_file = rels_dir / f"{xml_file.name}.rels" + + if not rels_file.exists(): + continue + + try: + rels_root = lxml.etree.parse(str(rels_file)).getroot() + rid_to_type = {} + + for rel in rels_root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ): + rid = rel.get("Id") + rel_type = rel.get("Type", "") + if rid: + if rid in rid_to_type: + rels_rel_path = rels_file.relative_to(self.unpacked_dir) + errors.append( + f" {rels_rel_path}: Line {rel.sourceline}: " + f"Duplicate relationship ID '{rid}' (IDs must be unique)" + ) + type_name = ( + rel_type.split("/")[-1] if "/" in rel_type else rel_type + ) + rid_to_type[rid] = type_name + + xml_root = lxml.etree.parse(str(xml_file)).getroot() + + r_ns = self.OFFICE_RELATIONSHIPS_NAMESPACE + rid_attrs_to_check = ["id", "embed", "link"] + for elem in xml_root.iter(): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + for attr_name in rid_attrs_to_check: + rid_attr = elem.get(f"{{{r_ns}}}{attr_name}") + if not rid_attr: + continue + xml_rel_path = xml_file.relative_to(self.unpacked_dir) + elem_name = ( + elem.tag.split("}")[-1] if "}" in elem.tag else elem.tag + ) + + if rid_attr not in rid_to_type: + errors.append( + f" {xml_rel_path}: Line {elem.sourceline}: " + f"<{elem_name}> r:{attr_name} references non-existent relationship '{rid_attr}' " + f"(valid IDs: {', '.join(sorted(rid_to_type.keys())[:5])}{'...' if len(rid_to_type) > 5 else ''})" + ) + elif attr_name == "id" and self.ELEMENT_RELATIONSHIP_TYPES: + expected_type = self._get_expected_relationship_type( + elem_name + ) + if expected_type: + actual_type = rid_to_type[rid_attr] + if expected_type not in actual_type.lower(): + errors.append( + f" {xml_rel_path}: Line {elem.sourceline}: " + f"<{elem_name}> references '{rid_attr}' which points to '{actual_type}' " + f"but should point to a '{expected_type}' relationship" + ) + + except Exception as e: + xml_rel_path = xml_file.relative_to(self.unpacked_dir) + errors.append(f" Error processing {xml_rel_path}: {e}") + + if errors: + print(f"FAILED - Found {len(errors)} relationship ID reference errors:") + for error in errors: + print(error) + print("\nThese ID mismatches will cause the document to appear corrupt!") + return False + else: + if self.verbose: + print("PASSED - All relationship ID references are valid") + return True + + def _get_expected_relationship_type(self, element_name): + elem_lower = element_name.lower() + + if elem_lower in self.ELEMENT_RELATIONSHIP_TYPES: + return self.ELEMENT_RELATIONSHIP_TYPES[elem_lower] + + if elem_lower.endswith("id") and len(elem_lower) > 2: + prefix = elem_lower[:-2] + if prefix.endswith("master"): + return prefix.lower() + elif prefix.endswith("layout"): + return prefix.lower() + else: + if prefix == "sld": + return "slide" + return prefix.lower() + + if elem_lower.endswith("reference") and len(elem_lower) > 9: + prefix = elem_lower[:-9] + return prefix.lower() + + return None + + def validate_content_types(self): + errors = [] + + content_types_file = self.unpacked_dir / "[Content_Types].xml" + if not content_types_file.exists(): + print("FAILED - [Content_Types].xml file not found") + return False + + try: + root = lxml.etree.parse(str(content_types_file)).getroot() + declared_parts = set() + declared_extensions = set() + + for override in root.findall( + f".//{{{self.CONTENT_TYPES_NAMESPACE}}}Override" + ): + part_name = override.get("PartName") + if part_name is not None: + declared_parts.add(part_name.lstrip("/")) + + for default in root.findall( + f".//{{{self.CONTENT_TYPES_NAMESPACE}}}Default" + ): + extension = default.get("Extension") + if extension is not None: + declared_extensions.add(extension.lower()) + + declarable_roots = { + "sld", + "sldLayout", + "sldMaster", + "presentation", + "document", + "workbook", + "worksheet", + "theme", + } + + media_extensions = { + "png": "image/png", + "jpg": "image/jpeg", + "jpeg": "image/jpeg", + "gif": "image/gif", + "bmp": "image/bmp", + "tiff": "image/tiff", + "wmf": "image/x-wmf", + "emf": "image/x-emf", + } + + all_files = list(self.unpacked_dir.rglob("*")) + all_files = [f for f in all_files if f.is_file()] + + for xml_file in self.xml_files: + path_str = str(xml_file.relative_to(self.unpacked_dir)).replace( + "\\", "/" + ) + + if any( + skip in path_str + for skip in [".rels", "[Content_Types]", "docProps/", "_rels/"] + ): + continue + + try: + root_tag = lxml.etree.parse(str(xml_file)).getroot().tag + root_name = root_tag.split("}")[-1] if "}" in root_tag else root_tag + + if root_name in declarable_roots and path_str not in declared_parts: + errors.append( + f" {path_str}: File with <{root_name}> root not declared in [Content_Types].xml" + ) + + except Exception: + continue + + for file_path in all_files: + if file_path.suffix.lower() in {".xml", ".rels"}: + continue + if file_path.name == "[Content_Types].xml": + continue + if "_rels" in file_path.parts or "docProps" in file_path.parts: + continue + + extension = file_path.suffix.lstrip(".").lower() + if extension and extension not in declared_extensions: + if extension in media_extensions: + relative_path = file_path.relative_to(self.unpacked_dir) + errors.append( + f' {relative_path}: File with extension \'{extension}\' not declared in [Content_Types].xml - should add: ' + ) + + except Exception as e: + errors.append(f" Error parsing [Content_Types].xml: {e}") + + if errors: + print(f"FAILED - Found {len(errors)} content type declaration errors:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print( + "PASSED - All content files are properly declared in [Content_Types].xml" + ) + return True + + def validate_file_against_xsd(self, xml_file, verbose=False): + xml_file = Path(xml_file).resolve() + unpacked_dir = self.unpacked_dir.resolve() + + is_valid, current_errors = self._validate_single_file_xsd( + xml_file, unpacked_dir + ) + + if is_valid is None: + return None, set() + elif is_valid: + return True, set() + + original_errors = self._get_original_file_errors(xml_file) + + assert current_errors is not None + new_errors = current_errors - original_errors + + new_errors = { + e for e in new_errors + if not any(pattern in e for pattern in self.IGNORED_VALIDATION_ERRORS) + } + + if new_errors: + if verbose: + relative_path = xml_file.relative_to(unpacked_dir) + print(f"FAILED - {relative_path}: {len(new_errors)} new error(s)") + for error in list(new_errors)[:3]: + truncated = error[:250] + "..." if len(error) > 250 else error + print(f" - {truncated}") + return False, new_errors + else: + if verbose: + print( + f"PASSED - No new errors (original had {len(current_errors)} errors)" + ) + return True, set() + + def validate_against_xsd(self): + new_errors = [] + original_error_count = 0 + valid_count = 0 + skipped_count = 0 + + for xml_file in self.xml_files: + relative_path = str(xml_file.relative_to(self.unpacked_dir)) + is_valid, new_file_errors = self.validate_file_against_xsd( + xml_file, verbose=False + ) + + if is_valid is None: + skipped_count += 1 + continue + elif is_valid and not new_file_errors: + valid_count += 1 + continue + elif is_valid: + original_error_count += 1 + valid_count += 1 + continue + + new_errors.append(f" {relative_path}: {len(new_file_errors)} new error(s)") + for error in list(new_file_errors)[:3]: + new_errors.append( + f" - {error[:250]}..." if len(error) > 250 else f" - {error}" + ) + + if self.verbose: + print(f"Validated {len(self.xml_files)} files:") + print(f" - Valid: {valid_count}") + print(f" - Skipped (no schema): {skipped_count}") + if original_error_count: + print(f" - With original errors (ignored): {original_error_count}") + print( + f" - With NEW errors: {len(new_errors) > 0 and len([e for e in new_errors if not e.startswith(' ')]) or 0}" + ) + + if new_errors: + print("\nFAILED - Found NEW validation errors:") + for error in new_errors: + print(error) + return False + else: + if self.verbose: + print("\nPASSED - No new XSD validation errors introduced") + return True + + def _get_schema_path(self, xml_file): + if xml_file.name in self.SCHEMA_MAPPINGS: + return self.schemas_dir / self.SCHEMA_MAPPINGS[xml_file.name] + + if xml_file.suffix == ".rels": + return self.schemas_dir / self.SCHEMA_MAPPINGS[".rels"] + + if "charts/" in str(xml_file) and xml_file.name.startswith("chart"): + return self.schemas_dir / self.SCHEMA_MAPPINGS["chart"] + + if "theme/" in str(xml_file) and xml_file.name.startswith("theme"): + return self.schemas_dir / self.SCHEMA_MAPPINGS["theme"] + + if xml_file.parent.name in self.MAIN_CONTENT_FOLDERS: + return self.schemas_dir / self.SCHEMA_MAPPINGS[xml_file.parent.name] + + return None + + def _clean_ignorable_namespaces(self, xml_doc): + xml_string = lxml.etree.tostring(xml_doc, encoding="unicode") + xml_copy = lxml.etree.fromstring(xml_string) + + for elem in xml_copy.iter(): + attrs_to_remove = [] + + for attr in elem.attrib: + if "{" in attr: + ns = attr.split("}")[0][1:] + if ns not in self.OOXML_NAMESPACES: + attrs_to_remove.append(attr) + + for attr in attrs_to_remove: + del elem.attrib[attr] + + self._remove_ignorable_elements(xml_copy) + + return lxml.etree.ElementTree(xml_copy) + + def _remove_ignorable_elements(self, root): + elements_to_remove = [] + + for elem in list(root): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + + tag_str = str(elem.tag) + if tag_str.startswith("{"): + ns = tag_str.split("}")[0][1:] + if ns not in self.OOXML_NAMESPACES: + elements_to_remove.append(elem) + continue + + self._remove_ignorable_elements(elem) + + for elem in elements_to_remove: + root.remove(elem) + + def _preprocess_for_mc_ignorable(self, xml_doc): + root = xml_doc.getroot() + + if f"{{{self.MC_NAMESPACE}}}Ignorable" in root.attrib: + del root.attrib[f"{{{self.MC_NAMESPACE}}}Ignorable"] + + return xml_doc + + def _preprocess_for_schema(self, xml_doc, relative_path): + return xml_doc + + def _validate_single_file_xsd(self, xml_file, base_path, schema_path=None): + schema_path = schema_path or self._get_schema_path(xml_file) + if not schema_path: + return None, None + + try: + schema = _load_schema(str(schema_path)) + + with open(xml_file, "r") as f: + xml_doc = lxml.etree.parse(f) + + xml_doc, _ = self._remove_template_tags_from_text_nodes(xml_doc) + xml_doc = self._preprocess_for_mc_ignorable(xml_doc) + + relative_path = xml_file.relative_to(base_path) + if ( + relative_path.parts + and relative_path.parts[0] in self.MAIN_CONTENT_FOLDERS + ): + xml_doc = self._clean_ignorable_namespaces(xml_doc) + + xml_doc = self._preprocess_for_schema(xml_doc, relative_path) + + if schema.validate(xml_doc): + return True, set() + else: + errors = set() + for error in schema.error_log: + errors.add(error.message) + return False, errors + + except Exception as e: + return False, {str(e)} + + def _get_original_file_errors(self, xml_file, schema_path=None): + if self.original_file is None: + return set() + + import tempfile + import zipfile + + xml_file = Path(xml_file).resolve() + unpacked_dir = self.unpacked_dir.resolve() + relative_path = xml_file.relative_to(unpacked_dir) + + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + + try: + with zipfile.ZipFile(self.original_file, "r") as zip_ref: + safe_extract(zip_ref, temp_path) + except (zipfile.BadZipFile, ValueError, OSError): + return set() + + original_xml_file = temp_path / relative_path + + if not original_xml_file.exists(): + return set() + + is_valid, errors = self._validate_single_file_xsd( + original_xml_file, temp_path, schema_path=schema_path + ) + return errors if errors else set() + + def _remove_template_tags_from_text_nodes(self, xml_doc): + warnings = [] + template_pattern = re.compile(r"\{\{[^}]*\}\}") + + xml_string = lxml.etree.tostring(xml_doc, encoding="unicode") + xml_copy = lxml.etree.fromstring(xml_string) + + def process_text_content(text, content_type): + if not text: + return text + matches = list(template_pattern.finditer(text)) + if matches: + for match in matches: + warnings.append( + f"Found template tag in {content_type}: {match.group()}" + ) + return template_pattern.sub("", text) + return text + + for elem in xml_copy.iter(): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + tag_str = str(elem.tag) + if tag_str.endswith("}t") or tag_str == "t": + continue + + elem.text = process_text_content(elem.text, "text content") + elem.tail = process_text_content(elem.tail, "tail content") + + return lxml.etree.ElementTree(xml_copy), warnings + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-pptx/scripts/office/validators/docx.py b/.github/skills/anthropic-pptx/scripts/office/validators/docx.py new file mode 100644 index 00000000..b1814994 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/validators/docx.py @@ -0,0 +1,466 @@ +""" +Validator for Word document XML files against XSD schemas. +""" + +import random +import re +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.minidom +import lxml.etree + +from helpers import safe_extract + +from .base import BaseSchemaValidator + + +class DOCXSchemaValidator(BaseSchemaValidator): + + WORD_2006_NAMESPACE = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + W14_NAMESPACE = "http://schemas.microsoft.com/office/word/2010/wordml" + W16CID_NAMESPACE = "http://schemas.microsoft.com/office/word/2016/wordml/cid" + + ELEMENT_RELATIONSHIP_TYPES = {} + + def validate(self): + if not self.validate_xml(): + return False + + all_valid = True + if not self.validate_namespaces(): + all_valid = False + + if not self.validate_unique_ids(): + all_valid = False + + if not self.validate_file_references(): + all_valid = False + + if not self.validate_content_types(): + all_valid = False + + if not self.validate_against_xsd(): + all_valid = False + + if not self.validate_whitespace_preservation(): + all_valid = False + + if not self.validate_deletions(): + all_valid = False + + if not self.validate_insertions(): + all_valid = False + + if not self.validate_all_relationship_ids(): + all_valid = False + + if not self.validate_id_constraints(): + all_valid = False + + if not self.validate_comment_markers(): + all_valid = False + + self.compare_paragraph_counts() + + return all_valid + + def validate_whitespace_preservation(self): + errors = [] + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + + for elem in root.iter(f"{{{self.WORD_2006_NAMESPACE}}}t"): + if elem.text: + text = elem.text + if re.search(r"^[ \t\n\r]", text) or re.search( + r"[ \t\n\r]$", text + ): + xml_space_attr = f"{{{self.XML_NAMESPACE}}}space" + if ( + xml_space_attr not in elem.attrib + or elem.attrib[xml_space_attr] != "preserve" + ): + text_preview = ( + repr(text)[:50] + "..." + if len(repr(text)) > 50 + else repr(text) + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: w:t element with whitespace missing xml:space='preserve': {text_preview}" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} whitespace preservation violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All whitespace is properly preserved") + return True + + def validate_deletions(self): + errors = [] + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + namespaces = {"w": self.WORD_2006_NAMESPACE} + + for t_elem in root.xpath(".//w:del//w:t", namespaces=namespaces): + if t_elem.text: + text_preview = ( + repr(t_elem.text)[:50] + "..." + if len(repr(t_elem.text)) > 50 + else repr(t_elem.text) + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {t_elem.sourceline}: found within : {text_preview}" + ) + + for instr_elem in root.xpath( + ".//w:del//w:instrText", namespaces=namespaces + ): + text_preview = ( + repr(instr_elem.text or "")[:50] + "..." + if len(repr(instr_elem.text or "")) > 50 + else repr(instr_elem.text or "") + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {instr_elem.sourceline}: found within (use ): {text_preview}" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} deletion validation violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - No w:t elements found within w:del elements") + return True + + def count_paragraphs_in_unpacked(self): + count = 0 + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + paragraphs = root.findall(f".//{{{self.WORD_2006_NAMESPACE}}}p") + count = len(paragraphs) + except Exception as e: + print(f"Error counting paragraphs in unpacked document: {e}") + + return count + + def count_paragraphs_in_original(self): + original = self.original_file + if original is None: + return 0 + + count = 0 + + try: + with tempfile.TemporaryDirectory() as temp_dir: + with zipfile.ZipFile(original, "r") as zip_ref: + safe_extract(zip_ref, Path(temp_dir)) + + doc_xml_path = temp_dir + "/word/document.xml" + root = lxml.etree.parse(doc_xml_path).getroot() + + paragraphs = root.findall(f".//{{{self.WORD_2006_NAMESPACE}}}p") + count = len(paragraphs) + + except Exception as e: + print(f"Error counting paragraphs in original document: {e}") + + return count + + def validate_insertions(self): + errors = [] + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + namespaces = {"w": self.WORD_2006_NAMESPACE} + + invalid_elements = root.xpath( + ".//w:ins//w:delText[not(ancestor::w:del)]", namespaces=namespaces + ) + + for elem in invalid_elements: + text_preview = ( + repr(elem.text or "")[:50] + "..." + if len(repr(elem.text or "")) > 50 + else repr(elem.text or "") + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: within : {text_preview}" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} insertion validation violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - No w:delText elements within w:ins elements") + return True + + def compare_paragraph_counts(self): + new_count = self.count_paragraphs_in_unpacked() + if self.original_file is None: + print(f"\nParagraphs: {new_count}") + return + + original_count = self.count_paragraphs_in_original() + diff = new_count - original_count + diff_str = f"+{diff}" if diff > 0 else str(diff) + print(f"\nParagraphs: {original_count} → {new_count} ({diff_str})") + + def _parse_id_value(self, val: str, base: int = 16) -> int: + return int(val, base) + + def validate_id_constraints(self): + errors = [] + para_id_attr = f"{{{self.W14_NAMESPACE}}}paraId" + durable_id_attr = f"{{{self.W16CID_NAMESPACE}}}durableId" + + for xml_file in self.xml_files: + try: + for elem in lxml.etree.parse(str(xml_file)).iter(): + if val := elem.get(para_id_attr): + try: + if self._parse_id_value(val, base=16) >= 0x80000000: + errors.append( + f" {xml_file.name}:{elem.sourceline}: paraId={val} >= 0x80000000" + ) + except ValueError: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"paraId={val} is not valid hex" + ) + + if val := elem.get(durable_id_attr): + if xml_file.name == "numbering.xml": + try: + if self._parse_id_value(val, base=10) >= 0x7FFFFFFF: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} >= 0x7FFFFFFF" + ) + except ValueError: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} must be decimal in numbering.xml" + ) + else: + try: + if self._parse_id_value(val, base=16) >= 0x7FFFFFFF: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} >= 0x7FFFFFFF" + ) + except ValueError: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} is not valid hex" + ) + except lxml.etree.XMLSyntaxError: + continue + + if errors: + print(f"FAILED - {len(errors)} ID constraint violations:") + for e in errors: + print(e) + elif self.verbose: + print("PASSED - All paraId/durableId values within constraints") + return not errors + + def validate_comment_markers(self): + errors = [] + + document_xml = None + comments_xml = None + for xml_file in self.xml_files: + if xml_file.name == "document.xml" and "word" in str(xml_file): + document_xml = xml_file + elif xml_file.name == "comments.xml": + comments_xml = xml_file + + if not document_xml: + if self.verbose: + print("PASSED - No document.xml found (skipping comment validation)") + return True + + try: + doc_root = lxml.etree.parse(str(document_xml)).getroot() + namespaces = {"w": self.WORD_2006_NAMESPACE} + + range_starts = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in doc_root.xpath( + ".//w:commentRangeStart", namespaces=namespaces + ) + } + range_ends = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in doc_root.xpath( + ".//w:commentRangeEnd", namespaces=namespaces + ) + } + references = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in doc_root.xpath( + ".//w:commentReference", namespaces=namespaces + ) + } + + orphaned_ends = range_ends - range_starts + for comment_id in sorted( + orphaned_ends, key=lambda x: int(x) if x and x.isdigit() else 0 + ): + errors.append( + f' document.xml: commentRangeEnd id="{comment_id}" has no matching commentRangeStart' + ) + + orphaned_starts = range_starts - range_ends + for comment_id in sorted( + orphaned_starts, key=lambda x: int(x) if x and x.isdigit() else 0 + ): + errors.append( + f' document.xml: commentRangeStart id="{comment_id}" has no matching commentRangeEnd' + ) + + comment_ids = set() + if comments_xml and comments_xml.exists(): + comments_root = lxml.etree.parse(str(comments_xml)).getroot() + comment_ids = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in comments_root.xpath( + ".//w:comment", namespaces=namespaces + ) + } + + marker_ids = range_starts | range_ends | references + invalid_refs = marker_ids - comment_ids + for comment_id in sorted( + invalid_refs, key=lambda x: int(x) if x and x.isdigit() else 0 + ): + if comment_id: + errors.append( + f' document.xml: marker id="{comment_id}" references non-existent comment' + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append(f" Error parsing XML: {e}") + + if errors: + print(f"FAILED - {len(errors)} comment marker violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All comment markers properly paired") + return True + + def repair(self) -> int: + repairs = super().repair() + repairs += self.repair_durableId() + return repairs + + def repair_durableId(self) -> int: + DURABLE_ID_ATTRS = ("w16cid:durableId", "w16cex:durableId") + repairs = 0 + renames: dict = {} + + for xml_file in self.xml_files: + try: + content = xml_file.read_text(encoding="utf-8") + dom = defusedxml.minidom.parseString(content) + is_numbering = xml_file.name == "numbering.xml" + base = 10 if is_numbering else 16 + pending = [] + seen_in_file = set() + modified = False + + for elem in dom.getElementsByTagName("*"): + for attr_name in DURABLE_ID_ATTRS: + if not elem.hasAttribute(attr_name): + continue + + durable_id = elem.getAttribute(attr_name) + try: + key = self._parse_id_value(durable_id, base=base) + needs_repair = key >= 0x7FFFFFFF + except ValueError: + key = durable_id + needs_repair = True + + if needs_repair: + if key in seen_in_file: + value = random.randint(1, 0x7FFFFFFE) + else: + seen_in_file.add(key) + if key not in renames: + renames[key] = random.randint(1, 0x7FFFFFFE) + value = renames[key] + new_id = str(value) if is_numbering else f"{value:08X}" + + elem.setAttribute(attr_name, new_id) + pending.append( + f" Repaired: {xml_file.name}: durableId {durable_id} → {new_id}" + ) + modified = True + + if modified: + xml_file.write_bytes(dom.toxml(encoding="UTF-8")) + for message in pending: + print(message) + repairs += len(pending) + + except Exception: + pass + + return repairs + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-pptx/scripts/office/validators/pptx.py b/.github/skills/anthropic-pptx/scripts/office/validators/pptx.py new file mode 100644 index 00000000..318f0e61 --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/validators/pptx.py @@ -0,0 +1,441 @@ +""" +Validator for PowerPoint presentation XML files against XSD schemas. +""" + +import re +from pathlib import Path + +from helpers import opc_target, rels_source_part, safe_extract + +from .base import BaseSchemaValidator + + +class PPTXSchemaValidator(BaseSchemaValidator): + + PRESENTATIONML_NAMESPACE = ( + "http://schemas.openxmlformats.org/presentationml/2006/main" + ) + + ELEMENT_RELATIONSHIP_TYPES = { + "sldid": "slide", + "sldmasterid": "slidemaster", + "notesmasterid": "notesmaster", + "sldlayoutid": "slidelayout", + "themeid": "theme", + "tablestyleid": "tablestyles", + } + + def validate(self): + if not self.validate_xml(): + return False + + all_valid = True + if not self.validate_namespaces(): + all_valid = False + + if not self.validate_unique_ids(): + all_valid = False + + if not self.validate_uuid_ids(): + all_valid = False + + if not self.validate_file_references(): + all_valid = False + + if not self.validate_slide_layout_ids(): + all_valid = False + + if not self.validate_content_types(): + all_valid = False + + if not self.validate_against_xsd(): + all_valid = False + + if not self.validate_notes_slide_references(): + all_valid = False + + if not self.validate_all_relationship_ids(): + all_valid = False + + if not self.validate_no_duplicate_slide_layouts(): + all_valid = False + + if not self.validate_master_theme_uniqueness(): + all_valid = False + + if not self.validate_charts(): + all_valid = False + + if not self.validate_slides(): + all_valid = False + + return all_valid + + def _package_map(self) -> dict: + wanted = [] + wanted += list(self.unpacked_dir.glob("[[]Content_Types[]].xml")) + wanted += list(self.unpacked_dir.glob("ppt/presentation.xml")) + wanted += list(self.unpacked_dir.glob("ppt/theme/*.xml")) + wanted += list(self.unpacked_dir.glob("ppt/theme/_rels/*.rels")) + wanted += list(self.unpacked_dir.glob("ppt/charts/chart*.xml")) + for group in ("slideMasters", "notesMasters", "handoutMasters"): + wanted += list(self.unpacked_dir.glob(f"ppt/{group}/*.xml")) + wanted += list(self.unpacked_dir.glob(f"ppt/{group}/_rels/*.rels")) + return { + p.relative_to(self.unpacked_dir).as_posix(): p.read_bytes() + for p in wanted + if p.is_file() + } + + def validate_master_theme_uniqueness(self): + from helpers.pptx_theme import _NOTES_MASTERS, live_shared_master_themes + + shared = live_shared_master_themes(self._package_map()) + if shared: + print(f"FAILED - Found {len(shared)} master(s) sharing a theme part:") + for message in shared: + print(f" {message}") + if any(m.startswith(_NOTES_MASTERS) for m in shared): + print(" Fix: in ppt/presentation.xml, move back to " + "directly after . PowerPoint reads that happily.") + else: + print(" Fix: give each master its own theme part.") + return False + + if self.verbose: + print("PASSED - No master shares a theme part in a way PowerPoint refuses") + return True + + def validate_charts(self): + from helpers.pptx_chart import find_chart_problems + + problems = find_chart_problems(self._package_map()) + if problems: + print(f"FAILED - Found {len(problems)} chart problem(s) PowerPoint rejects:") + for message in problems: + print(f" {message}") + return False + + if self.verbose: + print("PASSED - Charts satisfy the constraints PowerPoint enforces") + return True + + def _original_slide_defects(self, schema) -> set[str]: + import tempfile + import zipfile + + from helpers.pptx_slide import SLIDE_PART_RE, fatal_slide_errors + + if self.original_file is None: + return set() + + found: set[str] = set() + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + try: + with zipfile.ZipFile(self.original_file, "r") as zf: + safe_extract(zf, temp_path) + except (zipfile.BadZipFile, ValueError, OSError): + return set() + + for part in sorted(temp_path.rglob("*.xml")): + relative = part.relative_to(temp_path).as_posix() + if not SLIDE_PART_RE.fullmatch(relative): + continue + ok, errors = self._validate_single_file_xsd( + part.resolve(), temp_path.resolve(), schema_path=schema + ) + if ok is None or ok or not errors: + continue + found |= set(fatal_slide_errors(set(errors))) + return found + + def validate_slides(self): + from helpers.pptx_slide import ( + SLIDE_PART_RE, + fatal_slide_errors, + is_schema_verdict, + ) + + schema = self.schemas_dir / self.SCHEMA_MAPPINGS["ppt"] + inherited = self._original_slide_defects(schema) + problems: list[str] = [] + broken: list[str] = [] + + for xml_file in self.xml_files: + relative = xml_file.relative_to(self.unpacked_dir).as_posix() + if not SLIDE_PART_RE.fullmatch(relative): + continue + ok, errors = self._validate_single_file_xsd( + xml_file.resolve(), self.unpacked_dir.resolve(), schema_path=schema + ) + if ok is None or not errors: + continue + + unreadable = [f"{relative}: {e}" for e in errors if not is_schema_verdict(e)] + if unreadable: + broken.extend(unreadable) + continue + if ok: + continue + + for message in fatal_slide_errors(set(errors)): + if message in inherited: + continue + problems.append(f"{relative}: {message}") + + if broken: + print(f"FAILED - Could not check {len(broken)} slide part(s):") + for message in sorted(broken): + print(f" {message[:240]}") + + if problems: + print(f"FAILED - Found {len(problems)} slide problem(s) PowerPoint rejects:") + for message in sorted(problems): + print(f" {message[:240]}") + + if broken or problems: + return False + + if self.verbose: + print("PASSED - Slide XML has none of the defects PowerPoint refuses") + return True + + def _get_schema_path(self, xml_file): + if xml_file.parent.name == "charts" and xml_file.name.startswith("chart"): + return None + return super()._get_schema_path(xml_file) + + def _preprocess_for_schema(self, xml_doc, relative_path): + if relative_path.as_posix() != "ppt/presentation.xml": + return xml_doc + + root = xml_doc.getroot() + ns = f"{{{self.PRESENTATIONML_NAMESPACE}}}" + notes = root.find(f"{ns}notesMasterIdLst") + slides = root.find(f"{ns}sldIdLst") + if notes is None or slides is None: + return xml_doc + + children = list(root) + if children.index(notes) < children.index(slides): + return xml_doc + + root.remove(notes) + root.insert(list(root).index(slides), notes) + return xml_doc + + def validate_uuid_ids(self): + import lxml.etree + + errors = [] + uuid_pattern = re.compile( + r"^[\{\(]?[0-9A-Fa-f]{8}-?[0-9A-Fa-f]{4}-?[0-9A-Fa-f]{4}-?[0-9A-Fa-f]{4}-?[0-9A-Fa-f]{12}[\}\)]?$" + ) + + for xml_file in self.xml_files: + try: + root = lxml.etree.parse(str(xml_file)).getroot() + + for elem in root.iter(): + for attr, value in elem.attrib.items(): + attr_name = attr.split("}")[-1].lower() + if attr_name == "id" or attr_name.endswith("id"): + if self._looks_like_uuid(value): + if not uuid_pattern.match(value): + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: ID '{value}' appears to be a UUID but contains invalid hex characters" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} UUID ID validation errors:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All UUID-like IDs contain valid hex values") + return True + + def _looks_like_uuid(self, value): + clean_value = value.strip("{}()").replace("-", "") + return len(clean_value) == 32 and all(c.isalnum() for c in clean_value) + + def validate_slide_layout_ids(self): + import lxml.etree + + errors = [] + + slide_masters = list(self.unpacked_dir.glob("ppt/slideMasters/*.xml")) + + if not slide_masters: + if self.verbose: + print("PASSED - No slide masters found") + return True + + for slide_master in slide_masters: + try: + root = lxml.etree.parse(str(slide_master)).getroot() + + rels_file = slide_master.parent / "_rels" / f"{slide_master.name}.rels" + + if not rels_file.exists(): + errors.append( + f" {slide_master.relative_to(self.unpacked_dir)}: " + f"Missing relationships file: {rels_file.relative_to(self.unpacked_dir)}" + ) + continue + + rels_root = lxml.etree.parse(str(rels_file)).getroot() + + valid_layout_rids = set() + for rel in rels_root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ): + rel_type = rel.get("Type", "") + if "slideLayout" in rel_type: + valid_layout_rids.add(rel.get("Id")) + + for sld_layout_id in root.findall( + f".//{{{self.PRESENTATIONML_NAMESPACE}}}sldLayoutId" + ): + r_id = sld_layout_id.get( + f"{{{self.OFFICE_RELATIONSHIPS_NAMESPACE}}}id" + ) + layout_id = sld_layout_id.get("id") + + if r_id and r_id not in valid_layout_rids: + errors.append( + f" {slide_master.relative_to(self.unpacked_dir)}: " + f"Line {sld_layout_id.sourceline}: sldLayoutId with id='{layout_id}' " + f"references r:id='{r_id}' which is not found in slide layout relationships" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {slide_master.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} slide layout ID validation errors:") + for error in errors: + print(error) + print( + "Remove invalid references or add missing slide layouts to the relationships file." + ) + return False + else: + if self.verbose: + print("PASSED - All slide layout IDs reference valid slide layouts") + return True + + def validate_no_duplicate_slide_layouts(self): + import lxml.etree + + errors = [] + slide_rels_files = list(self.unpacked_dir.glob("ppt/slides/_rels/*.xml.rels")) + + for rels_file in slide_rels_files: + try: + root = lxml.etree.parse(str(rels_file)).getroot() + + layout_rels = [ + rel + for rel in root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ) + if "slideLayout" in rel.get("Type", "") + ] + + if len(layout_rels) > 1: + errors.append( + f" {rels_file.relative_to(self.unpacked_dir)}: has {len(layout_rels)} slideLayout references" + ) + + except Exception as e: + errors.append( + f" {rels_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print("FAILED - Found slides with duplicate slideLayout references:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All slides have exactly one slideLayout reference") + return True + + def validate_notes_slide_references(self): + import lxml.etree + + errors = [] + notes_slide_references = {} + + slide_rels_files = list(self.unpacked_dir.glob("ppt/slides/_rels/*.xml.rels")) + + if not slide_rels_files: + if self.verbose: + print("PASSED - No slide relationship files found") + return True + + for rels_file in slide_rels_files: + try: + root = lxml.etree.parse(str(rels_file)).getroot() + + for rel in root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ): + rel_type = rel.get("Type", "") + if "notesSlide" in rel_type: + part = opc_target( + rel.get("Target", ""), + rels_source_part(rels_file, self.unpacked_dir), + rel.get("TargetMode", ""), + ) + if part: + slide_name = rels_file.stem.replace( + ".xml", "" + ) + + notes_slide_references.setdefault(part, []).append( + (slide_name, rels_file) + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {rels_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + for target, references in notes_slide_references.items(): + if len(references) > 1: + slide_names = [ref[0] for ref in references] + errors.append( + f" Notes slide '{target}' is referenced by multiple slides: {', '.join(slide_names)}" + ) + for slide_name, rels_file in references: + errors.append(f" - {rels_file.relative_to(self.unpacked_dir)}") + + if errors: + print( + f"FAILED - Found {len([e for e in errors if not e.startswith(' ')])} notes slide reference validation errors:" + ) + for error in errors: + print(error) + print("Each slide may optionally have its own slide file.") + return False + else: + if self.verbose: + print("PASSED - All notes slide references are unique") + return True + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-pptx/scripts/office/validators/redlining.py b/.github/skills/anthropic-pptx/scripts/office/validators/redlining.py new file mode 100644 index 00000000..4185c51f --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/office/validators/redlining.py @@ -0,0 +1,299 @@ +""" +Validator for tracked changes in Word documents. + +Detects untracked edits in word/document.xml: text that differs from the +original without a / wrapper recording it. The tracked changes +that are new relative to the original are undone, and the result is compared +against the original; whatever text still differs was edited without being +tracked. + +Only the document body is compared. Headers, footers, footnotes and endnotes +are separate parts and are not checked. +""" + +import subprocess +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.ElementTree as ET +from defusedxml.common import DefusedXmlException + +from helpers import rendered_text, safe_extract + + +class RedliningValidator: + + def __init__(self, unpacked_dir, original_docx, verbose=False): + self.unpacked_dir = Path(unpacked_dir) + self.original_docx = Path(original_docx) + self.verbose = verbose + self.namespaces = { + "w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + } + + def repair(self) -> int: + return 0 + + def validate(self): + modified_file = self.unpacked_dir / "word" / "document.xml" + if not modified_file.exists(): + print(f"FAILED - Modified document.xml not found at {modified_file}") + return False + + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + + try: + with zipfile.ZipFile(self.original_docx, "r") as zip_ref: + safe_extract(zip_ref, temp_path) + except Exception as e: + print(f"FAILED - Error unpacking original docx: {e}") + return False + + original_file = temp_path / "word" / "document.xml" + if not original_file.exists(): + print( + f"FAILED - Original document.xml not found in {self.original_docx}" + ) + return False + + try: + modified_tree = ET.parse(modified_file) + modified_root = modified_tree.getroot() + original_tree = ET.parse(original_file) + original_root = original_tree.getroot() + except (ET.ParseError, DefusedXmlException) as e: + print(f"FAILED - Error parsing XML files: {e}") + return False + + new_changes = self._new_tracked_changes(original_root, modified_root) + self._remove_tracked_changes(modified_root, new_changes) + + modified_text = self._extract_text_content(modified_root) + original_text = self._extract_text_content(original_root) + + if modified_text != original_text: + error_message = self._generate_detailed_diff( + original_text, modified_text + ) + print(error_message) + return False + + if self.verbose: + print( + f"PASSED - All {len(new_changes)} change(s) against the original " + "are properly tracked" + ) + return True + + def _tracked_change_elements(self, root): + ins_tag = f"{{{self.namespaces['w']}}}ins" + del_tag = f"{{{self.namespaces['w']}}}del" + return [elem for elem in root.iter() if elem.tag in (ins_tag, del_tag)] + + def _rendered_text(self, elem): + preserve = elem.get("{http://www.w3.org/XML/1998/namespace}space") == "preserve" + return rendered_text(elem.text or "", preserve) + + def _text_elements(self, elem): + w = self.namespaces["w"] + return [ + node + for node in elem.iter() + if node.tag in (f"{{{w}}}t", f"{{{w}}}delText") + ] + + def _tracked_change_key(self, elem): + w = self.namespaces["w"] + text = "".join(self._rendered_text(node) for node in self._text_elements(elem)) + return (elem.tag, elem.get(f"{{{w}}}author"), elem.get(f"{{{w}}}date"), text) + + def _new_tracked_changes(self, original_root, modified_root): + original = self._tracked_change_elements(original_root) + modified = self._tracked_change_elements(modified_root) + + pool = {} + for elem in original: + pool.setdefault(self._tracked_change_key(elem), []).append(elem) + + matched, leftover = set(), [] + for elem in modified: + bucket = pool.get(self._tracked_change_key(elem)) + if bucket: + matched.add(bucket.pop()) + else: + leftover.append(elem) + + def group(elem): + return self._tracked_change_key(elem)[:3] + + def text_of(elems): + return "".join(self._tracked_change_key(e)[3] for e in elems) + + unmatched_original = {} + for elem in original: + if elem not in matched: + unmatched_original.setdefault(group(elem), []).append(elem) + + by_group = {} + for elem in leftover: + by_group.setdefault(group(elem), []).append(elem) + + new = set() + for key, elems in by_group.items(): + rebuilt = text_of(elems) + if rebuilt and rebuilt == text_of(unmatched_original.get(key, [])): + continue + new.update(elems) + return new + + def _generate_detailed_diff(self, original_text, modified_text): + error_parts = [ + "FAILED - Document text doesn't match after removing the tracked changes", + "", + "Likely causes:", + " 1. Modified text inside another author's or tags", + " 2. Made edits without proper tracked changes", + " 3. Didn't nest inside when deleting another's insertion", + " 4. Rewrote another author's / and changed its text on", + " the way. A tracked change from the original is recognised by its", + " author, date and text; anything that doesn't reproduce one exactly", + " reads as new, and the text it carried is reported missing.", + "", + "For pre-redlined documents, use correct patterns:", + " - To reject another's INSERTION: Nest inside their ", + " - To reject PART of one: nest around only the runs you reject.", + " Their may be split around it, so long as the pieces keep", + " their author and date and still spell out the same text.", + " - To restore another's DELETION: Add new AFTER their ", + "", + ] + + git_diff = self._get_git_word_diff(original_text, modified_text) + if git_diff: + error_parts.extend(["Differences:", "============", git_diff]) + else: + error_parts.append("Unable to generate word diff (git not available)") + + return "\n".join(error_parts) + + def _get_git_word_diff(self, original_text, modified_text): + try: + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + + original_file = temp_path / "original.txt" + modified_file = temp_path / "modified.txt" + + original_file.write_text(original_text, encoding="utf-8") + modified_file.write_text(modified_text, encoding="utf-8") + + result = subprocess.run( + [ + "git", + "diff", + "--word-diff=plain", + "--word-diff-regex=.", + "-U0", + "--no-index", + str(original_file), + str(modified_file), + ], + capture_output=True, + text=True, + ) + + if result.stdout.strip(): + lines = result.stdout.split("\n") + content_lines = [] + in_content = False + for line in lines: + if line.startswith("@@"): + in_content = True + continue + if in_content and line.strip(): + content_lines.append(line) + + if content_lines: + return "\n".join(content_lines) + + result = subprocess.run( + [ + "git", + "diff", + "--word-diff=plain", + "-U0", + "--no-index", + str(original_file), + str(modified_file), + ], + capture_output=True, + text=True, + ) + + if result.stdout.strip(): + lines = result.stdout.split("\n") + content_lines = [] + in_content = False + for line in lines: + if line.startswith("@@"): + in_content = True + continue + if in_content and line.strip(): + content_lines.append(line) + return "\n".join(content_lines) + + except (subprocess.CalledProcessError, FileNotFoundError, Exception): + pass + + return None + + def _remove_tracked_changes(self, root, targets): + ins_tag = f"{{{self.namespaces['w']}}}ins" + del_tag = f"{{{self.namespaces['w']}}}del" + + for parent in root.iter(): + to_remove = [] + for child in parent: + if child.tag == ins_tag and child in targets: + to_remove.append(child) + for elem in to_remove: + parent.remove(elem) + + deltext_tag = f"{{{self.namespaces['w']}}}delText" + t_tag = f"{{{self.namespaces['w']}}}t" + + for parent in root.iter(): + to_process = [] + for child in parent: + if child.tag == del_tag and child in targets: + to_process.append((child, list(parent).index(child))) + + for del_elem, del_index in reversed(to_process): + for elem in del_elem.iter(): + if elem.tag == deltext_tag: + elem.tag = t_tag + + for child in reversed(list(del_elem)): + parent.insert(del_index, child) + parent.remove(del_elem) + + def _extract_text_content(self, root): + p_tag = f"{{{self.namespaces['w']}}}p" + t_tag = f"{{{self.namespaces['w']}}}t" + + paragraphs = [] + for p_elem in root.findall(f".//{p_tag}"): + text_parts = [] + for t_elem in p_elem.findall(f".//{t_tag}"): + text_parts.append(self._rendered_text(t_elem)) + paragraph_text = "".join(text_parts) + if paragraph_text: + paragraphs.append(paragraph_text) + + return "\n".join(paragraphs) + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-pptx/scripts/thumbnail.py b/.github/skills/anthropic-pptx/scripts/thumbnail.py new file mode 100644 index 00000000..d49cac0e --- /dev/null +++ b/.github/skills/anthropic-pptx/scripts/thumbnail.py @@ -0,0 +1,311 @@ +"""Create thumbnail grids from PowerPoint presentation slides. + +Creates a grid layout of slide thumbnails for quick visual analysis. +Labels each thumbnail with its XML filename (e.g., slide1.xml). +Hidden slides are shown with a placeholder pattern. + +Usage: + python thumbnail.py input.pptx [output_prefix] [--cols N] + +Examples: + python thumbnail.py presentation.pptx + # Creates: thumbnails.jpg + + python thumbnail.py template.pptx grid --cols 4 + # Creates: grid.jpg (or grid-1.jpg, grid-2.jpg for large decks) +""" + +import argparse +import posixpath +import subprocess +import sys +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.minidom +from defusedxml import ElementTree +from office.helpers import SLIDE_REL_TYPE, opc_target +from office.soffice import run_soffice +from PIL import Image, ImageDraw, ImageFont + + +THUMBNAIL_WIDTH = 300 +CONVERSION_DPI = 100 +MAX_COLS = 6 +DEFAULT_COLS = 3 +JPEG_QUALITY = 95 +GRID_PADDING = 20 +BORDER_WIDTH = 2 +FONT_SIZE_RATIO = 0.10 +LABEL_PADDING_RATIO = 0.4 + + +def main(): + parser = argparse.ArgumentParser( + description="Create thumbnail grids from PowerPoint slides." + ) + parser.add_argument("input", help="Input PowerPoint file (.pptx)") + parser.add_argument( + "output_prefix", + nargs="?", + default="thumbnails", + help="Output prefix for image files (default: thumbnails)", + ) + parser.add_argument( + "--cols", + type=int, + default=DEFAULT_COLS, + help=f"Number of columns (default: {DEFAULT_COLS}, max: {MAX_COLS})", + ) + + args = parser.parse_args() + + cols = min(args.cols, MAX_COLS) + if args.cols > MAX_COLS: + print(f"Warning: Columns limited to {MAX_COLS}") + + input_path = Path(args.input) + if not input_path.exists() or input_path.suffix.lower() != ".pptx": + print(f"Error: Invalid PowerPoint file: {args.input}", file=sys.stderr) + sys.exit(1) + + output_path = Path(f"{args.output_prefix}.jpg") + + try: + slide_info = get_slide_info(input_path) + + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + visible_images = convert_to_images(input_path, temp_path) + + if not visible_images and not any(s["hidden"] for s in slide_info): + print("Error: No slides found", file=sys.stderr) + sys.exit(1) + + slides = build_slide_list(slide_info, visible_images, temp_path) + + grid_files = create_grids(slides, cols, THUMBNAIL_WIDTH, output_path) + + print(f"Created {len(grid_files)} grid(s):") + for grid_file in grid_files: + print(f" {grid_file}") + + except Exception as e: + print(f"Error: {e}", file=sys.stderr) + sys.exit(1) + + +def _is_hidden(zf: zipfile.ZipFile, part: str) -> bool: + try: + with zf.open(part) as f: + for _, root in ElementTree.iterparse(f, events=("start",)): + return root.get("show") in ("0", "false") + except (KeyError, ElementTree.ParseError): + return False + return False + + +def get_slide_info(pptx_path: Path) -> list[dict]: + with zipfile.ZipFile(pptx_path, "r") as zf: + rels_content = zf.read("ppt/_rels/presentation.xml.rels").decode("utf-8") + rels_dom = defusedxml.minidom.parseString(rels_content) + + rid_to_part = {} + for rel in rels_dom.getElementsByTagName("Relationship"): + if rel.getAttribute("Type") != SLIDE_REL_TYPE: + continue + part = opc_target( + rel.getAttribute("Target"), + "ppt/presentation.xml", + rel.getAttribute("TargetMode"), + ) + if part is not None: + rid_to_part[rel.getAttribute("Id")] = part + + pres_content = zf.read("ppt/presentation.xml").decode("utf-8") + pres_dom = defusedxml.minidom.parseString(pres_content) + + present = set(zf.namelist()) + + slides = [] + for sld_id in pres_dom.getElementsByTagName("p:sldId"): + part = rid_to_part.get(sld_id.getAttribute("r:id")) + if part is not None and part in present: + slides.append( + {"name": posixpath.basename(part), "hidden": _is_hidden(zf, part)} + ) + + return slides + + +def build_slide_list( + slide_info: list[dict], + visible_images: list[Path], + temp_dir: Path, +) -> list[tuple[Path, str]]: + visible_count = sum(1 for info in slide_info if not info["hidden"]) + rendered_hidden = len(visible_images) == len(slide_info) != visible_count + + if not rendered_hidden and visible_count != len(visible_images): + raise ValueError( + f"LibreOffice rendered {len(visible_images)} page(s) for {visible_count} " + f"visible slide(s) of {len(slide_info)}; thumbnails would be mislabeled" + ) + + if visible_images: + with Image.open(visible_images[0]) as img: + placeholder_size = img.size + else: + placeholder_size = (1920, 1080) + + slides = [] + visible_idx = 0 + + for info in slide_info: + if info["hidden"] and not rendered_hidden: + placeholder_path = temp_dir / f"hidden-{info['name']}.jpg" + placeholder_img = create_hidden_placeholder(placeholder_size) + placeholder_img.save(placeholder_path, "JPEG") + slides.append((placeholder_path, f"{info['name']} (hidden)")) + else: + label = f"{info['name']} (hidden)" if info["hidden"] else info["name"] + slides.append((visible_images[visible_idx], label)) + visible_idx += 1 + + return slides + + +def create_hidden_placeholder(size: tuple[int, int]) -> Image.Image: + img = Image.new("RGB", size, color="#F0F0F0") + draw = ImageDraw.Draw(img) + line_width = max(5, min(size) // 100) + draw.line([(0, 0), size], fill="#CCCCCC", width=line_width) + draw.line([(size[0], 0), (0, size[1])], fill="#CCCCCC", width=line_width) + return img + + +def convert_to_images(pptx_path: Path, temp_dir: Path) -> list[Path]: + pdf_path = temp_dir / f"{pptx_path.stem}.pdf" + + result = run_soffice( + ["--headless", "--convert-to", "pdf", "--outdir", str(temp_dir), str(pptx_path)], + capture_output=True, + text=True, + ) + if result.returncode != 0 or not pdf_path.exists(): + detail = (result.stderr or result.stdout or "").strip() + raise RuntimeError(f"PDF conversion failed: {detail}" if detail else "PDF conversion failed") + + result = subprocess.run( + [ + "pdftoppm", + "-jpeg", + "-r", + str(CONVERSION_DPI), + str(pdf_path), + str(temp_dir / "slide"), + ], + capture_output=True, + text=True, + ) + if result.returncode != 0: + raise RuntimeError("Image conversion failed") + + return sorted(temp_dir.glob("slide-*.jpg")) + + +def create_grids( + slides: list[tuple[Path, str]], + cols: int, + width: int, + output_path: Path, +) -> list[str]: + max_per_grid = cols * (cols + 1) + grid_files = [] + + for chunk_idx, start_idx in enumerate(range(0, len(slides), max_per_grid)): + end_idx = min(start_idx + max_per_grid, len(slides)) + chunk_slides = slides[start_idx:end_idx] + + grid = create_grid(chunk_slides, cols, width) + + if len(slides) <= max_per_grid: + grid_filename = output_path + else: + stem = output_path.stem + suffix = output_path.suffix + grid_filename = output_path.parent / f"{stem}-{chunk_idx + 1}{suffix}" + + grid_filename.parent.mkdir(parents=True, exist_ok=True) + grid.save(str(grid_filename), quality=JPEG_QUALITY) + grid_files.append(str(grid_filename)) + + return grid_files + + +def create_grid( + slides: list[tuple[Path, str]], + cols: int, + width: int, +) -> Image.Image: + font_size = int(width * FONT_SIZE_RATIO) + label_padding = int(font_size * LABEL_PADDING_RATIO) + + with Image.open(slides[0][0]) as img: + aspect = img.height / img.width + height = int(width * aspect) + + rows = (len(slides) + cols - 1) // cols + grid_w = cols * width + (cols + 1) * GRID_PADDING + grid_h = rows * (height + font_size + label_padding * 2) + (rows + 1) * GRID_PADDING + + grid = Image.new("RGB", (grid_w, grid_h), "white") + draw = ImageDraw.Draw(grid) + + try: + font = ImageFont.load_default(size=font_size) + except Exception: + font = ImageFont.load_default() + + for i, (img_path, slide_name) in enumerate(slides): + row, col = i // cols, i % cols + x = col * width + (col + 1) * GRID_PADDING + y_base = ( + row * (height + font_size + label_padding * 2) + (row + 1) * GRID_PADDING + ) + + label = slide_name + bbox = draw.textbbox((0, 0), label, font=font) + text_w = bbox[2] - bbox[0] + draw.text( + (x + (width - text_w) // 2, y_base + label_padding), + label, + fill="black", + font=font, + ) + + y_thumbnail = y_base + label_padding + font_size + label_padding + + with Image.open(img_path) as img: + img.thumbnail((width, height), Image.Resampling.LANCZOS) + w, h = img.size + tx = x + (width - w) // 2 + ty = y_thumbnail + (height - h) // 2 + grid.paste(img, (tx, ty)) + + if BORDER_WIDTH > 0: + draw.rectangle( + [ + (tx - BORDER_WIDTH, ty - BORDER_WIDTH), + (tx + w + BORDER_WIDTH - 1, ty + h + BORDER_WIDTH - 1), + ], + outline="gray", + width=BORDER_WIDTH, + ) + + return grid + + +if __name__ == "__main__": + main() diff --git a/.github/skills/anthropic-xlsx/LICENSE.txt b/.github/skills/anthropic-xlsx/LICENSE.txt new file mode 100644 index 00000000..c55ab422 --- /dev/null +++ b/.github/skills/anthropic-xlsx/LICENSE.txt @@ -0,0 +1,30 @@ +© 2025 Anthropic, PBC. All rights reserved. + +LICENSE: Use of these materials (including all code, prompts, assets, files, +and other components of this Skill) is governed by your agreement with +Anthropic regarding use of Anthropic's services. If no separate agreement +exists, use is governed by Anthropic's Consumer Terms of Service or +Commercial Terms of Service, as applicable: +https://www.anthropic.com/legal/consumer-terms +https://www.anthropic.com/legal/commercial-terms +Your applicable agreement is referred to as the "Agreement." "Services" are +as defined in the Agreement. + +ADDITIONAL RESTRICTIONS: Notwithstanding anything in the Agreement to the +contrary, users may not: + +- Extract these materials from the Services or retain copies of these + materials outside the Services +- Reproduce or copy these materials, except for temporary copies created + automatically during authorized use of the Services +- Create derivative works based on these materials +- Distribute, sublicense, or transfer these materials to any third party +- Make, offer to sell, sell, or import any inventions embodied in these + materials +- Reverse engineer, decompile, or disassemble these materials + +The receipt, viewing, or possession of these materials does not convey or +imply any license or right beyond those expressly granted above. + +Anthropic retains all right, title, and interest in these materials, +including all copyrights, patents, and other intellectual property rights. diff --git a/.github/skills/anthropic-xlsx/SKILL.md b/.github/skills/anthropic-xlsx/SKILL.md new file mode 100644 index 00000000..e8986975 --- /dev/null +++ b/.github/skills/anthropic-xlsx/SKILL.md @@ -0,0 +1,99 @@ +--- +name: anthropic-xlsx +description: "Use this skill any time a spreadsheet file is the primary input or output. This means any task where the user wants to: open, read, edit, or fix an existing .xlsx, .xlsm, .xltx, .csv, or .tsv file (e.g., adding columns, computing formulas, formatting, charting, cleaning messy data); create a new spreadsheet from scratch or from other data sources; or convert between tabular file formats. Trigger especially when the user references a spreadsheet file by name or path — even casually (like \"the xlsx in my downloads\") — and wants something done to it or produced from it. Also trigger for cleaning or restructuring messy tabular data files (malformed rows, misplaced headers, junk data) into proper spreadsheets. The deliverable must be a spreadsheet file. Do NOT trigger when the primary deliverable is a Word document, HTML report, standalone Python script, database pipeline, or Google Sheets API integration, even if tabular data is involved." +license: Proprietary. LICENSE.txt has complete terms +--- + +# XLSX creation, editing, and analysis + +| Task | Approach | +|---|---| +| **Create** or **edit** with formulas/formatting | `openpyxl` — see gotchas below | +| **Bulk data** in or out | `pandas` (`read_excel`, `to_excel`) | +| **Quick look** at a sheet | `markitdown file.xlsx` — `## SheetName` per sheet; reads `.xlsm` too. No cell coordinates, so don't plan edits from it | +| **Read** a model (formulas *and* values) | two `load_workbook` passes — see gotchas | + +> `openpyxl`, `pandas`, and `markitdown` are preinstalled — do not run `pip install` first; write the script and import directly. Only if an import fails (or the `markitdown` command is missing): `pip install` the missing package. + +> Script paths below are relative to this skill's directory. + +## Requirements for every output + +- **Professional font** (Arial, Times New Roman) throughout, unless the user says otherwise. +- **Zero formula errors.** Never ship while `recalc.py` reports `errors_found`. If you think an error predates you, prove it: load the *original* with `data_only=True` and look at that cell. An error you introduced looks exactly like one you inherited. +- **Use formulas, never hardcoded results.** Write `sheet['B10'] = '=SUM(B2:B9)'`, not the Python-computed total. The sheet must recalculate when its inputs change. +- **Follow the user's spec literally.** Exact tab names, exact column headers, and the formula they spelled out. A redesign that computes something else fails, however elegant. +- **Document every assumption and hardcoded number** where the reader will see it — a cell comment, or an adjacent cell at a table's end. Cite a real source when one exists (`Source: Company 10-K, FY2024, Page 45, Revenue Note, [SEC EDGAR URL]`); when the number came from the user, say so plainly. +- **A workbook *you create* for someone to fill in** needs a short legend naming which cells to edit, and one example row of realistic values showing the expected format. Never add such a row to a file you were asked to edit. +- **Editing an existing file: match its conventions exactly.** They override every guideline here. Find its designated input cells first — a distinct font color, fill, or shading marks them — write only there, and leave every existing formula untouched. + +## Recalculate (mandatory whenever the file contains formulas) + +openpyxl writes formulas as strings with **no cached values**. Until you recalculate, every +formula cell reads back as `None` to anything reading cached values — `pandas`, +`load_workbook(data_only=True)`, and most previewers. + +```bash +python scripts/recalc.py output.xlsx [timeout_seconds] # default 30 +``` + +LibreOffice computes every formula, the file is **rewritten in place**, and you get JSON: +`status` (`success` | `errors_found`), `total_formulas`, `total_errors`, and an +`error_summary` naming up to 100 cells per error type (`locations_truncated` says how many it +withheld — trust `total_errors`, not the length of the list). Fix what it names and run it +again. **JSON with an `error` key instead of a `status` means nothing was recalculated**, and +only that case exits non-zero — `errors_found` exits 0, so never treat a clean exit as a clean +workbook. + +**A green recalc proves your formulas *evaluate*, not that they are *right*.** An off-by-one +range or a reference to the wrong row yields a clean, error-free file with wrong numbers. +Write 2–3 formulas first and check they pull the values you expect, before building out a grid. + +**A workbook that links to another file loses those links** if you re-save it with openpyxl and +then recalculate. Such a formula reads `='[1]Returns Analysis'!$B$2` — the `[1]` is an index +into the workbook's external-reference list, naming a *separate file on disk*, not a sheet. +That file is rarely present here, so the cell's cached value is the only thing holding its +data. openpyxl strips that value on save; LibreOffice then has to resolve the reference for +real, fails, writes `#NAME?`, and deletes every link. `recalc.py` refuses to run in that state +— copy those cells' values out of the original before you save over them (`--force` overrides, +and accepts the loss). + +## Choosing formulas that survive verification + +LibreOffice implements fewer functions than Excel, and one it cannot evaluate becomes a +literal `#NAME?` baked into the file you deliver. + +- **Prefer Excel-2007-era functions** — `SUMIFS`, `INDEX`, `MATCH`, `IFERROR`, `SUMPRODUCT` — which need no prefix. +- **Six post-2007 functions work, but only with an `_xlfn.` prefix**, because openpyxl writes your formula into the XML verbatim and Excel stores post-2007 names prefixed (its UI hides the prefix): `_xlfn.TEXTJOIN`, `_xlfn.CONCAT`, `_xlfn.IFS`, `_xlfn.SWITCH`, `_xlfn.MAXIFS`, `_xlfn.MINIFS`. Written bare, each yields `#NAME?`. +- **Never use `XLOOKUP`, `XMATCH`, `SORT`, `FILTER`, `UNIQUE`, or `SEQUENCE`.** The runtime's LibreOffice cannot evaluate them under *any* prefix. Newer builds do evaluate them, but they are spilling array functions and an openpyxl-written file has no spill metadata, so only the top-left cell of the range gets a value — and `recalc.py` reports `total_errors: 0` on the truncated result. Use `INDEX`/`MATCH` for lookups, and sort, filter, and de-duplicate in Python before writing the cells. +- A formula LibreOffice could not parse is written back **lowercased** — a quick tell beside a `#NAME?`. + +## openpyxl gotchas + +- **Reading a model takes two loads.** `data_only=True` yields cached values with the formulas gone; the default yields formula strings with no values. One pass cannot give you both. +- **`data_only=True` is destructive if you save.** That workbook has no formulas left, so saving replaces every one with a literal — permanently. +- **`data_only=True` on a file openpyxl just wrote returns `None` everywhere** — run `recalc.py` first. (A formula whose result is `""` also reads back as `None`.) +- **Merged cells: write the top-left anchor only.** Every other cell in the range is a `MergedCell` whose `.value` is read-only. +- **`.xlsm` loses its macros unless you pass `keep_vba=True`** to `load_workbook`. +- **A sheet name containing a space must be quoted** in a cross-sheet reference: `='Assumptions Inputs'!$B$5`. Unquoted, it evaluates to `#VALUE!`. + +## Financial models + +Unless the user says otherwise, or the existing file already does something else. + +**Color:** blue text (`0,0,255`) for hardcoded inputs and scenario levers · black for formulas · +green (`0,128,0`) for links to another sheet · red (`255,0,0`) for links to another file · +yellow fill (`255,255,0`) for key assumptions and cells the user should fill in. + +**Numbers:** currency `$#,##0`, with the unit named in the header (`Revenue ($mm)`) · zeros +render as `-`, including in percentages (`$#,##0;($#,##0);-`) · negatives in parentheses · +percentages `0.0%`, **stored as fractions** (`0.15` renders `15.0%`; storing `15` renders +`1500.0%`) · valuation multiples `0.0x` · years as text (`"2024"`, never `2,024`). + +**Structure:** every assumption in its own labeled cell, referenced by the formulas that use it +(`=B5*(1+$B$6)`, never `=B5*1.05`) · formulas consistent across every projection period, since a +lone edited cell mid-row is the commonest silent error · guard denominators that can be zero. + +## Dependencies + +`openpyxl`, `pandas`, `markitdown` (pip, preinstalled — install only if an import fails or the command is missing) · LibreOffice (`soffice`, auto-configured for sandboxed environments via `scripts/office/soffice.py`) diff --git a/.github/skills/anthropic-xlsx/scripts/office/helpers/__init__.py b/.github/skills/anthropic-xlsx/scripts/office/helpers/__init__.py new file mode 100644 index 00000000..d3c5817c --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/helpers/__init__.py @@ -0,0 +1,111 @@ +import os +import posixpath +import re +import stat +import tempfile +import urllib.parse +import zipfile +from pathlib import Path + +OOXML_FAMILY = { + ".docx": "docx", + ".dotx": "docx", + ".pptx": "pptx", + ".potx": "pptx", + ".xlsx": "xlsx", + ".xltx": "xlsx", +} + +_SCHEME_RE = re.compile(r"^[A-Za-z][A-Za-z0-9+.\-]*:") + +SLIDE_REL_TYPE = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/slide" + + +def opc_target(target: str, source_part: str, target_mode: str = "") -> str | None: + if not target: + return None + if target_mode.lower() == "external": + return None + if _SCHEME_RE.match(target): + return None + + target = urllib.parse.unquote(target) + + if "\\" in target: + raise ValueError(f"relationship target is not a POSIX part name: {target!r}") + + if target.startswith("/"): + joined = target.lstrip("/") + else: + joined = posixpath.join(posixpath.dirname(source_part), target) + + parts: list[str] = [] + for segment in posixpath.normpath(joined).split("/"): + if segment in ("", "."): + continue + if segment == "..": + if not parts: + raise ValueError(f"relationship target escapes the package: {target!r}") + parts.pop() + else: + parts.append(segment) + + if not parts: + raise ValueError(f"relationship target resolves to nothing: {target!r}") + return "/".join(parts) + + +def rels_source_part(rels_file: Path, unpacked_dir: Path) -> str: + owner_dir = rels_file.parent.parent.relative_to(unpacked_dir) + return posixpath.join(owner_dir.as_posix(), rels_file.name[: -len(".rels")]).lstrip("./") + + +def part_text(data: bytes) -> str: + return data.decode("utf-8", "surrogateescape") + + +XML_SPACE = " \t\r\n" + + +def rendered_text(text: str, preserve: bool) -> str: + return text if preserve else text.strip(XML_SPACE) + + +def safe_extract(zf: zipfile.ZipFile, dest: Path) -> None: + dest = dest.resolve() + for m in zf.infolist(): + if stat.S_ISLNK(m.external_attr >> 16): + raise ValueError(f"symlink archive entry not allowed: {m.filename!r}") + target = (dest / m.filename).resolve() + if not target.is_relative_to(dest): + raise ValueError(f"unsafe archive entry: {m.filename!r}") + zf.extract(m, dest) + + +def rezip(src_dir: Path, out_path: Path) -> None: + files = sorted(p for p in src_dir.rglob("*") if p.is_file()) + ct = src_dir / "[Content_Types].xml" + fd, tmp_name = tempfile.mkstemp( + prefix=out_path.name + ".", suffix=".tmp", dir=out_path.parent + ) + tmp_out = Path(tmp_name) + try: + with os.fdopen(fd, "wb") as fh: + with zipfile.ZipFile(fh, "w", zipfile.ZIP_DEFLATED) as zf: + if ct.exists(): + zf.write(ct, ct.relative_to(src_dir), compress_type=zipfile.ZIP_STORED) + for f in files: + if f == ct: + continue + zf.write(f, f.relative_to(src_dir)) + if out_path.exists(): + mode = out_path.stat().st_mode & 0o777 + else: + umask = os.umask(0) + os.umask(umask) + mode = 0o666 & ~umask + os.chmod(tmp_out, mode) + os.replace(tmp_out, out_path) + finally: + if tmp_out.exists(): + tmp_out.unlink() diff --git a/.github/skills/anthropic-xlsx/scripts/office/helpers/pptx_chart.py b/.github/skills/anthropic-xlsx/scripts/office/helpers/pptx_chart.py new file mode 100644 index 00000000..209cb7c5 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/helpers/pptx_chart.py @@ -0,0 +1,170 @@ +"""Find chart XML that PowerPoint refuses but the schema accepts. + +Detection only: for either fault more than one repair is valid, and only the +author knows which was meant. +""" + + +from __future__ import annotations + +import re +from typing import Mapping + +from . import part_text + + +_CHART_PART_RE = re.compile(r"ppt/charts/chart\d+\.xml") + +_GROUPING_RE = re.compile(r"""]*?\bval=["'](\w+)["']""") +_DLBL_POS_RE = re.compile(r"""]*?\bval=["'](\w+)["']""") + +def _strip_ext_lst(text: str) -> str: + out, cursor = [], 0 + for lo, hi in _ext_lst_spans(text): + out.append(text[cursor:lo]) + cursor = hi + out.append(text[cursor:]) + return "".join(out) + +_BAR_GROUP_RE = re.compile(r"]*(?.*?", re.DOTALL) + +STACKED_GROUPINGS = frozenset({"stacked", "percentStacked"}) +ILLEGAL_ON_STACKED = frozenset({"outEnd"}) +LEGAL_ON_STACKED = ("ctr", "inEnd", "inBase") + + +def _check_stacked_label_positions(part: str, xml: str) -> list[str]: + problems: list[str] = [] + for match in _BAR_GROUP_RE.finditer(xml): + block = _strip_ext_lst(match.group(0)) + group = match.group(1) + + grouping = _GROUPING_RE.search(block) + if grouping is None or grouping.group(1) not in STACKED_GROUPINGS: + continue + + bad = [p for p in _DLBL_POS_RE.findall(block) if p in ILLEGAL_ON_STACKED] + for pos in sorted(set(bad)): + problems.append( + f'{part}: {bad.count(pos)} data label(s) use dLblPos="{pos}" on a ' + f"{grouping.group(1)} {group}; PowerPoint allows only " + f"{', '.join(LEGAL_ON_STACKED)} there" + ) + return problems + + + +_ANY_CHART_GROUP_RE = re.compile(r"]*(?.*?", re.DOTALL) + +_AXID_RE = re.compile( + r"""\s*]*?\bval=["'](-?\d+)["']\s*(?:/>|>\s*)""" +) + +_AXIS_DECL_RE = re.compile( + r"""]*(?\s*]*?\bval=["'](-?\d+)["']""" +) + +AXID_LIMIT = { + "barChart": 2, "lineChart": 2, "areaChart": 2, "scatterChart": 2, + "bubbleChart": 2, "radarChart": 2, "stockChart": 2, + "bar3DChart": 3, "line3DChart": 3, "area3DChart": 3, + "surfaceChart": 3, "surface3DChart": 3, +} + +AXID_MINIMUM = { + "barChart": 2, "lineChart": 2, "areaChart": 2, "scatterChart": 2, + "bubbleChart": 2, "radarChart": 2, "stockChart": 2, + "bar3DChart": 2, "area3DChart": 2, "surfaceChart": 2, + "line3DChart": 3, "surface3DChart": 3, +} + + +def _declared_axes(xml: str) -> dict[str, list[str]]: + axes: dict[str, list[str]] = {} + for kind, axid in _AXIS_DECL_RE.findall(xml): + axes.setdefault(kind, []).append(axid) + return axes + + +def _canonical_ids(axes: dict[str, list[str]], limit: int) -> list[str] | None: + category = axes.get("catAx", []) + axes.get("dateAx", []) + value = axes.get("valAx", []) + series = axes.get("serAx", []) + if len(category) != 1 or len(value) != 1 or len(series) > 1: + return None + ids = [category[0], value[0]] + if limit >= 3 and series: + ids.append(series[0]) + return ids + + +def _undeclared_axes(kind: str, block: str, axes: dict[str, list[str]]) -> list[str] | None: + if kind not in AXID_LIMIT: + return None + ids = _AXID_RE.findall(block) + declared = {i for group in axes.values() for i in group} + if len([i for i in ids if i in declared]) >= 2: + return None + return ids + + +def _check_chart_axis_references(part: str, xml: str) -> list[str]: + axes = _declared_axes(xml) + problems: list[str] = [] + declared = {i for group in axes.values() for i in group} + for match in _ANY_CHART_GROUP_RE.finditer(xml): + kind, block = match.group(1), match.group(0) + ids = _undeclared_axes(kind, block, axes) + if ids is None: + continue + if not ids: + problems.append( + f"{part}: declares no this part can resolve; a chart " + f"group needs {AXID_MINIMUM[kind]}, and PowerPoint discards one with fewer" + ) + continue + dead = [i for i in ids if i not in declared] + canonical = _canonical_ids(axes, AXID_LIMIT[kind]) + if canonical is not None and len(canonical) >= AXID_MINIMUM[kind]: + hint = f"Fix: point them at the axes this part declares ({', '.join(canonical)})" + else: + hint = ("Fix: the part declares several axes of a kind -- declare the " + "secondary axes the series expects, or drop them") + detail = (f"of which {', '.join(dead)} name no declared axis" + if dead else f"only {len(ids)} of which this part declares") + problems.append( + f"{part}: references axId {', '.join(ids)}, {detail}, " + f"leaving fewer than two live axes; PowerPoint discards the chart. {hint}" + ) + return problems + + +def _ext_lst_spans(text: str) -> list[tuple[int, int]]: + spans: list[tuple[int, int]] = [] + depth = 0 + start = 0 + for match in re.finditer(r"<(/?)c:extLst\b[^>]*?(/?)>", text): + closing, self_closing = match.group(1), match.group(2) + if self_closing: + continue + if closing: + depth -= 1 + if depth == 0: + spans.append((start, match.end())) + else: + if depth == 0: + start = match.start() + depth += 1 + return spans + + +CHART_CHECKS = (_check_stacked_label_positions, _check_chart_axis_references) + + +def find_chart_problems(files: Mapping[str, bytes]) -> list[str]: + problems: list[str] = [] + for part in sorted(n for n in files if _CHART_PART_RE.fullmatch(n)): + xml = part_text(files[part]) + for check in CHART_CHECKS: + problems.extend(check(part, xml)) + return problems diff --git a/.github/skills/anthropic-xlsx/scripts/office/helpers/pptx_slide.py b/.github/skills/anthropic-xlsx/scripts/office/helpers/pptx_slide.py new file mode 100644 index 00000000..22f9aee0 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/helpers/pptx_slide.py @@ -0,0 +1,60 @@ +"""Pick the slide-XML schema errors PowerPoint refuses the file over. + +A denylist over lxml's messages, so an unrecognised error class is a miss rather +than a false alarm. +""" + + +from __future__ import annotations + +import re + +SLIDE_PART_RE = re.compile( + r"ppt/(slides|slideLayouts|slideMasters|notesSlides|notesMasters|handoutMasters)" + r"/[^/]+\.xml" +) + +FATAL_SLIDE_ERRORS: tuple[tuple[re.Pattern[str], str], ...] = ( + ( + re.compile(r"\}tableStyleId': This element is not expected"), + "two in one (the schema allows one)", + ), + ( + re.compile(r"\}srgbClr', attribute 'val'"), + "a colour that is not six hex digits", + ), + ( + re.compile(r"\}txBody': Missing child element"), + "a with no children", + ), + ( + re.compile(r"\}miter', attribute 'lim'"), + 'a line join with lim="NaN"', + ), + ( + re.compile(r"\}uLnTx': This element is not expected"), + " in a position the schema forbids", + ), + ( + re.compile(r"\}overrideClrMapping': This element is not expected"), + " in a position the schema forbids", + ), + ( + re.compile(r"\}nvGrpSpPr': Missing child element"), + "a with no children", + ), +) + + +def is_schema_verdict(error: str) -> bool: + return error.startswith("Element ") + + +def fatal_slide_errors(errors: set[str]) -> list[str]: + out = [] + for error in sorted(errors): + for pattern, meaning in FATAL_SLIDE_ERRORS: + if pattern.search(error): + out.append(f"{meaning}: {error}") + break + return out diff --git a/.github/skills/anthropic-xlsx/scripts/office/helpers/pptx_theme.py b/.github/skills/anthropic-xlsx/scripts/office/helpers/pptx_theme.py new file mode 100644 index 00000000..84466201 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/helpers/pptx_theme.py @@ -0,0 +1,114 @@ +"""Find masters sharing a theme part in the way PowerPoint refuses to open. + +Reports only; the fix is to move back to directly after + in ppt/presentation.xml. +""" + + +from __future__ import annotations + +import posixpath +import re +from typing import Mapping + +from . import part_text + +THEME_REL_TYPE = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/theme" + +_MASTER_RE = re.compile( + r"^ppt/(?PslideMasters|notesMasters|handoutMasters)/" + r"(?:slide|notes|handout)Master(?P\d+)\.xml$" +) +_GROUP_ORDER = {"slideMasters": 0, "notesMasters": 1, "handoutMasters": 2} + +_RELATIONSHIP_RE = re.compile( + r"]*?(?:/>|>.*?)", re.DOTALL +) + + +def _sort_key(name: str) -> tuple[int, int]: + m = _MASTER_RE.match(name) + assert m is not None + return (_GROUP_ORDER[m.group("group")], int(m.group("num"))) + + +def _rels_path(part: str) -> str: + directory, base = posixpath.split(part) + return f"{directory}/_rels/{base}.rels" + + +def _resolve(rels_path: str, target: str) -> str: + if target.startswith("/"): + return target.lstrip("/") + part_dir = posixpath.dirname(posixpath.dirname(rels_path)) + return posixpath.normpath(posixpath.join(part_dir, target)) + + +def _theme_rel(files: Mapping[str, bytes], master: str): + rels_path = _rels_path(master) + rels = files.get(rels_path) + if rels is None: + return None + for element in _RELATIONSHIP_RE.findall(part_text(rels)): + if f'Type="{THEME_REL_TYPE}"' not in element: + continue + target = re.search(r'\bTarget="([^"]+)"', element) + if target is None: + continue + return rels_path, element, _resolve(rels_path, target.group(1)) + return None + + +def _masters(files: Mapping[str, bytes]) -> list[str]: + return sorted((n for n in files if _MASTER_RE.match(n)), key=_sort_key) + + +_PRESENTATION = "ppt/presentation.xml" +_NOTES_MASTERS = "ppt/notesMasters/" +_IGNORABLE_RE = re.compile(r"|<\?.*?\?>", re.DOTALL) +_AFTER_SLDIDLST_RE = re.compile( + r"]*/>|[^>]*>.*?)\s*(<[^>\s/]+)", re.DOTALL +) + + +def _notes_master_share_is_inert(files: Mapping[str, bytes]) -> bool: + data = files.get(_PRESENTATION) + if data is None: + return False + match = _AFTER_SLDIDLST_RE.search(_IGNORABLE_RE.sub("", part_text(data))) + return match is not None and match.group(1) == " bool: + return inert_notes and master.startswith(_NOTES_MASTERS) + + +def find_shared_master_themes(files: Mapping[str, bytes]) -> list[str]: + return [ + f"{master} shares {theme} with {first}" + for master, _, _, theme, first in _shares(files) + ] + + +def live_shared_master_themes(files: Mapping[str, bytes]) -> list[str]: + inert_notes = _notes_master_share_is_inert(files) + return [ + f"{master} shares {theme} with {first}" + for master, _, _, theme, first in _shares(files) + if not _is_inert(master, inert_notes) + ] diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chart.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chart.xsd new file mode 100644 index 00000000..6454ef9a --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chart.xsd @@ -0,0 +1,1499 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chartDrawing.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chartDrawing.xsd new file mode 100644 index 00000000..afa4f463 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-chartDrawing.xsd @@ -0,0 +1,146 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-diagram.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-diagram.xsd new file mode 100644 index 00000000..64e66b8a --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-diagram.xsd @@ -0,0 +1,1085 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-lockedCanvas.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-lockedCanvas.xsd new file mode 100644 index 00000000..687eea82 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-lockedCanvas.xsd @@ -0,0 +1,11 @@ + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-main.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-main.xsd new file mode 100644 index 00000000..6ac81b06 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-main.xsd @@ -0,0 +1,3081 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-picture.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-picture.xsd new file mode 100644 index 00000000..1dbf0514 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-picture.xsd @@ -0,0 +1,23 @@ + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-spreadsheetDrawing.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-spreadsheetDrawing.xsd new file mode 100644 index 00000000..f1af17db --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-spreadsheetDrawing.xsd @@ -0,0 +1,185 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-wordprocessingDrawing.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-wordprocessingDrawing.xsd new file mode 100644 index 00000000..0a185ab6 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/dml-wordprocessingDrawing.xsd @@ -0,0 +1,287 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/pml.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/pml.xsd new file mode 100644 index 00000000..14ef4888 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/pml.xsd @@ -0,0 +1,1676 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-additionalCharacteristics.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-additionalCharacteristics.xsd new file mode 100644 index 00000000..c20f3bf1 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-additionalCharacteristics.xsd @@ -0,0 +1,28 @@ + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-bibliography.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-bibliography.xsd new file mode 100644 index 00000000..ac602522 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-bibliography.xsd @@ -0,0 +1,144 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-commonSimpleTypes.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-commonSimpleTypes.xsd new file mode 100644 index 00000000..424b8ba8 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-commonSimpleTypes.xsd @@ -0,0 +1,174 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlDataProperties.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlDataProperties.xsd new file mode 100644 index 00000000..2bddce29 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlDataProperties.xsd @@ -0,0 +1,25 @@ + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlSchemaProperties.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlSchemaProperties.xsd new file mode 100644 index 00000000..8a8c18ba --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-customXmlSchemaProperties.xsd @@ -0,0 +1,18 @@ + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd new file mode 100644 index 00000000..5c42706a --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd @@ -0,0 +1,59 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd new file mode 100644 index 00000000..853c341c --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd @@ -0,0 +1,56 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesVariantTypes.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesVariantTypes.xsd new file mode 100644 index 00000000..da835ee8 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-documentPropertiesVariantTypes.xsd @@ -0,0 +1,195 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-math.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-math.xsd new file mode 100644 index 00000000..87ad2658 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-math.xsd @@ -0,0 +1,582 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-relationshipReference.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-relationshipReference.xsd new file mode 100644 index 00000000..9e86f1b2 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/shared-relationshipReference.xsd @@ -0,0 +1,25 @@ + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/sml.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/sml.xsd new file mode 100644 index 00000000..d0be42e7 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/sml.xsd @@ -0,0 +1,4439 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-main.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-main.xsd new file mode 100644 index 00000000..8821dd18 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-main.xsd @@ -0,0 +1,570 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-officeDrawing.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-officeDrawing.xsd new file mode 100644 index 00000000..ca2575c7 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-officeDrawing.xsd @@ -0,0 +1,509 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-presentationDrawing.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-presentationDrawing.xsd new file mode 100644 index 00000000..dd079e60 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-presentationDrawing.xsd @@ -0,0 +1,12 @@ + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-spreadsheetDrawing.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-spreadsheetDrawing.xsd new file mode 100644 index 00000000..3dd6cf62 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-spreadsheetDrawing.xsd @@ -0,0 +1,108 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-wordprocessingDrawing.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-wordprocessingDrawing.xsd new file mode 100644 index 00000000..f1041e34 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/vml-wordprocessingDrawing.xsd @@ -0,0 +1,96 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/wml.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/wml.xsd new file mode 100644 index 00000000..9c5b7a63 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/wml.xsd @@ -0,0 +1,3646 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/xml.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/xml.xsd new file mode 100644 index 00000000..0f13678d --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ISO-IEC29500-4_2016/xml.xsd @@ -0,0 +1,116 @@ + + + + + + See http://www.w3.org/XML/1998/namespace.html and + http://www.w3.org/TR/REC-xml for information about this namespace. + + This schema document describes the XML namespace, in a form + suitable for import by other schema documents. + + Note that local names in this namespace are intended to be defined + only by the World Wide Web Consortium or its subgroups. The + following names are currently defined in this namespace and should + not be used with conflicting semantics by any Working Group, + specification, or document instance: + + base (as an attribute name): denotes an attribute whose value + provides a URI to be used as the base for interpreting any + relative URIs in the scope of the element on which it + appears; its value is inherited. This name is reserved + by virtue of its definition in the XML Base specification. + + lang (as an attribute name): denotes an attribute whose value + is a language code for the natural language of the content of + any element; its value is inherited. This name is reserved + by virtue of its definition in the XML specification. + + space (as an attribute name): denotes an attribute whose + value is a keyword indicating what whitespace processing + discipline is intended for the content of the element; its + value is inherited. This name is reserved by virtue of its + definition in the XML specification. + + Father (in any context at all): denotes Jon Bosak, the chair of + the original XML Working Group. This name is reserved by + the following decision of the W3C XML Plenary and + XML Coordination groups: + + In appreciation for his vision, leadership and dedication + the W3C XML Plenary on this 10th day of February, 2000 + reserves for Jon Bosak in perpetuity the XML name + xml:Father + + + + + This schema defines attributes and an attribute group + suitable for use by + schemas wishing to allow xml:base, xml:lang or xml:space attributes + on elements they define. + + To enable this, such a schema must import this schema + for the XML namespace, e.g. as follows: + <schema . . .> + . . . + <import namespace="http://www.w3.org/XML/1998/namespace" + schemaLocation="http://www.w3.org/2001/03/xml.xsd"/> + + Subsequently, qualified reference to any of the attributes + or the group defined below will have the desired effect, e.g. + + <type . . .> + . . . + <attributeGroup ref="xml:specialAttrs"/> + + will define a type which will schema-validate an instance + element with any of those attributes + + + + In keeping with the XML Schema WG's standard versioning + policy, this schema document will persist at + http://www.w3.org/2001/03/xml.xsd. + At the date of issue it can also be found at + http://www.w3.org/2001/xml.xsd. + The schema document at that URI may however change in the future, + in order to remain compatible with the latest version of XML Schema + itself. In other words, if the XML Schema namespace changes, the version + of this document at + http://www.w3.org/2001/xml.xsd will change + accordingly; the version at + http://www.w3.org/2001/03/xml.xsd will not change. + + + + + + In due course, we should install the relevant ISO 2- and 3-letter + codes as the enumerated possible values . . . + + + + + + + + + + + + + + + See http://www.w3.org/TR/xmlbase/ for + information about this attribute. + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-contentTypes.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-contentTypes.xsd new file mode 100644 index 00000000..a6de9d27 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-contentTypes.xsd @@ -0,0 +1,42 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-coreProperties.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-coreProperties.xsd new file mode 100644 index 00000000..10e978b6 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-coreProperties.xsd @@ -0,0 +1,50 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-digSig.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-digSig.xsd new file mode 100644 index 00000000..4248bf7a --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-digSig.xsd @@ -0,0 +1,49 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-relationships.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-relationships.xsd new file mode 100644 index 00000000..56497467 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/ecma/fouth-edition/opc-relationships.xsd @@ -0,0 +1,33 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/mce/mc.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/mce/mc.xsd new file mode 100644 index 00000000..ef725457 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/mce/mc.xsd @@ -0,0 +1,75 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-2010.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-2010.xsd new file mode 100644 index 00000000..f65f7777 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-2010.xsd @@ -0,0 +1,560 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-2012.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-2012.xsd new file mode 100644 index 00000000..6b00755a --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-2012.xsd @@ -0,0 +1,67 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-2018.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-2018.xsd new file mode 100644 index 00000000..f321d333 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-2018.xsd @@ -0,0 +1,14 @@ + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-cex-2018.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-cex-2018.xsd new file mode 100644 index 00000000..364c6a9b --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-cex-2018.xsd @@ -0,0 +1,20 @@ + + + + + + + + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-cid-2016.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-cid-2016.xsd new file mode 100644 index 00000000..fed9d15b --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-cid-2016.xsd @@ -0,0 +1,13 @@ + + + + + + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-sdtdatahash-2020.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-sdtdatahash-2020.xsd new file mode 100644 index 00000000..680cf154 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-sdtdatahash-2020.xsd @@ -0,0 +1,4 @@ + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-symex-2015.xsd b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-symex-2015.xsd new file mode 100644 index 00000000..89ada908 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/schemas/microsoft/wml-symex-2015.xsd @@ -0,0 +1,8 @@ + + + + + + + + diff --git a/.github/skills/anthropic-xlsx/scripts/office/soffice.py b/.github/skills/anthropic-xlsx/scripts/office/soffice.py new file mode 100644 index 00000000..0b4c99de --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/soffice.py @@ -0,0 +1,192 @@ +""" +Helper for running LibreOffice (soffice) in environments where AF_UNIX +sockets may be blocked (e.g., sandboxed VMs). Detects the restriction +at runtime and applies an LD_PRELOAD shim if needed. + +Usage: + from office.soffice import run_soffice + + result = run_soffice(["--headless", "--convert-to", "pdf", "input.docx"]) + +Call soffice through run_soffice, not through subprocess with get_soffice_env(): +the env dict carries the shim but names no user profile, and a non-root sandbox +cannot bootstrap the default one -- soffice aborts with "User installation could +not be completed" and converts nothing. get_soffice_env() stays public for the +callers that build their own argv (they must pass -env:UserInstallation too). +""" + +import contextlib +import os +import socket +import subprocess +import tempfile +from collections.abc import Iterable +from pathlib import Path + + +def get_soffice_env() -> dict: + env = os.environ.copy() + env["SAL_USE_VCLPLUGIN"] = "svp" + + if _needs_shim(): + shim = _ensure_shim() + env["LD_PRELOAD"] = str(shim) + + return env + + +def run_soffice(args: Iterable[str], **kwargs) -> subprocess.CompletedProcess: + args = list(args) + with contextlib.ExitStack() as stack: + if not any(str(a).startswith("-env:UserInstallation") for a in args): + profile = stack.enter_context( + tempfile.TemporaryDirectory(prefix="lo_profile_", ignore_cleanup_errors=True) + ) + args = [f"-env:UserInstallation={Path(profile).as_uri()}"] + args + return subprocess.run(["soffice"] + args, env=get_soffice_env(), **kwargs) + + + +_SHIM_SO = Path(tempfile.gettempdir()) / "lo_socket_shim.so" + + +def _needs_shim() -> bool: + try: + s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + s.close() + return False + except OSError: + return True + + +def _ensure_shim() -> Path: + if _SHIM_SO.exists(): + return _SHIM_SO + + src = Path(tempfile.gettempdir()) / "lo_socket_shim.c" + src.write_text(_SHIM_SOURCE) + subprocess.run( + ["gcc", "-shared", "-fPIC", "-o", str(_SHIM_SO), str(src), "-ldl"], + check=True, + capture_output=True, + ) + src.unlink() + return _SHIM_SO + + + +_SHIM_SOURCE = r""" +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include + +static int (*real_socket)(int, int, int); +static int (*real_socketpair)(int, int, int, int[2]); +static int (*real_listen)(int, int); +static int (*real_accept)(int, struct sockaddr *, socklen_t *); +static int (*real_close)(int); +static int (*real_read)(int, void *, size_t); + +/* Per-FD bookkeeping (FDs >= 1024 are passed through unshimmed). */ +static int is_shimmed[1024]; +static int peer_of[1024]; +static int wake_r[1024]; /* accept() blocks reading this */ +static int wake_w[1024]; /* close() writes to this */ +static int listener_fd = -1; /* FD that received listen() */ + +__attribute__((constructor)) +static void init(void) { + real_socket = dlsym(RTLD_NEXT, "socket"); + real_socketpair = dlsym(RTLD_NEXT, "socketpair"); + real_listen = dlsym(RTLD_NEXT, "listen"); + real_accept = dlsym(RTLD_NEXT, "accept"); + real_close = dlsym(RTLD_NEXT, "close"); + real_read = dlsym(RTLD_NEXT, "read"); + for (int i = 0; i < 1024; i++) { + peer_of[i] = -1; + wake_r[i] = -1; + wake_w[i] = -1; + } +} + +/* ---- socket ---------------------------------------------------------- */ +int socket(int domain, int type, int protocol) { + if (domain == AF_UNIX) { + int fd = real_socket(domain, type, protocol); + if (fd >= 0) return fd; + /* socket(AF_UNIX) blocked – fall back to socketpair(). */ + int sv[2]; + if (real_socketpair(domain, type, protocol, sv) == 0) { + if (sv[0] >= 0 && sv[0] < 1024) { + is_shimmed[sv[0]] = 1; + peer_of[sv[0]] = sv[1]; + int wp[2]; + if (pipe(wp) == 0) { + wake_r[sv[0]] = wp[0]; + wake_w[sv[0]] = wp[1]; + } + } + return sv[0]; + } + errno = EPERM; + return -1; + } + return real_socket(domain, type, protocol); +} + +/* ---- listen ---------------------------------------------------------- */ +int listen(int sockfd, int backlog) { + if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) { + listener_fd = sockfd; + return 0; + } + return real_listen(sockfd, backlog); +} + +/* ---- accept ---------------------------------------------------------- */ +int accept(int sockfd, struct sockaddr *addr, socklen_t *addrlen) { + if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) { + /* Block until close() writes to the wake pipe. */ + if (wake_r[sockfd] >= 0) { + char buf; + real_read(wake_r[sockfd], &buf, 1); + } + errno = ECONNABORTED; + return -1; + } + return real_accept(sockfd, addr, addrlen); +} + +/* ---- close ----------------------------------------------------------- */ +int close(int fd) { + if (fd >= 0 && fd < 1024 && is_shimmed[fd]) { + int was_listener = (fd == listener_fd); + is_shimmed[fd] = 0; + + if (wake_w[fd] >= 0) { /* unblock accept() */ + char c = 0; + write(wake_w[fd], &c, 1); + real_close(wake_w[fd]); + wake_w[fd] = -1; + } + if (wake_r[fd] >= 0) { real_close(wake_r[fd]); wake_r[fd] = -1; } + if (peer_of[fd] >= 0) { real_close(peer_of[fd]); peer_of[fd] = -1; } + + if (was_listener) + _exit(0); /* conversion done – exit */ + } + return real_close(fd); +} +""" + + + +if __name__ == "__main__": + import sys + result = run_soffice(sys.argv[1:]) + sys.exit(result.returncode) diff --git a/.github/skills/anthropic-xlsx/scripts/office/validate.py b/.github/skills/anthropic-xlsx/scripts/office/validate.py new file mode 100644 index 00000000..29ca186a --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/validate.py @@ -0,0 +1,173 @@ +""" +Command line tool to validate Office document XML files against XSD schemas and tracked changes. + +Usage: + python validate.py [--original ] [--auto-repair] [--author NAME] + +The first argument can be either: +- An unpacked directory containing the Office document XML files +- A packed Office file (.docx/.pptx/.xlsx or .dotx/.potx/.xltx template) which will be unpacked to a temp directory + +Auto-repair fixes: +- paraId/durableId values that exceed OOXML limits +- Missing xml:space="preserve" on w:t elements with whitespace +""" + +import argparse +import sys +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.ElementTree as ET +from defusedxml.common import DefusedXmlException + +from helpers import OOXML_FAMILY, rezip, safe_extract +from validators import DOCXSchemaValidator, PPTXSchemaValidator, RedliningValidator + +WORD_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + + +def _fail(message: str): + print(f"Error: {message}", file=sys.stderr) + sys.exit(2) + + +def _has_tracked_changes(unpacked_dir: Path) -> bool: + document = unpacked_dir / "word" / "document.xml" + if not document.is_file(): + return False + try: + root = ET.parse(document).getroot() + except (ET.ParseError, DefusedXmlException): + return False + tracked = {f"{{{WORD_NS}}}ins", f"{{{WORD_NS}}}del"} + return any(elem.tag in tracked for elem in root.iter()) + + +def main(): + parser = argparse.ArgumentParser(description="Validate Office document XML files") + parser.add_argument( + "path", + help="Path to unpacked directory or packed Office file (.docx/.pptx/.xlsx or .dotx/.potx/.xltx)", + ) + parser.add_argument( + "--original", + required=False, + default=None, + help="Path to original file (.docx/.pptx/.xlsx or .dotx/.potx/.xltx). If omitted, all XSD errors are reported and redlining validation is skipped.", + ) + parser.add_argument( + "-v", + "--verbose", + action="store_true", + help="Enable verbose output", + ) + parser.add_argument( + "--auto-repair", + action="store_true", + help="Automatically repair common issues (hex IDs, whitespace preservation). " + "Modifies the input in place: repairs to a packed file are written back to it.", + ) + parser.add_argument( + "--author", + default=None, + help="The name you are redlining under. Passing it turns on the " + "tracked-change check: any text differing from --original without a " + "/ recording it is reported. Untracked edits carry no " + "author, so the check covers them whoever made them — the name marks " + "the run as redlining work and is not used to filter. Requires " + "--original; docx only.", + ) + args = parser.parse_args() + + if args.author is not None and not args.original: + _fail("--author requires --original") + + path = Path(args.path) + if not path.exists(): + _fail(f"{path} does not exist") + + original_file = None + if args.original: + original_file = Path(args.original) + if not original_file.is_file(): + _fail(f"{original_file} is not a file") + if original_file.suffix.lower() not in OOXML_FAMILY: + _fail(f"{original_file} must be one of: {', '.join(sorted(OOXML_FAMILY))}") + + family = OOXML_FAMILY.get((original_file or path).suffix.lower()) + if family is None: + _fail( + f"Cannot determine file type from {path}. Use --original or provide one of: {', '.join(sorted(OOXML_FAMILY))}." + ) + + if args.author is not None and family != "docx": + _fail(f"--author only applies to docx files, not {family}") + + packed_file = None + temp_dir_ctx = None + if path.is_file() and path.suffix.lower() in OOXML_FAMILY: + packed_file = path + temp_dir_ctx = tempfile.TemporaryDirectory() + unpacked_dir = Path(temp_dir_ctx.name) + try: + with zipfile.ZipFile(path, "r") as zf: + safe_extract(zf, unpacked_dir) + except (zipfile.BadZipFile, ValueError, OSError) as e: + _fail(f"cannot unpack {path}: {e}") + else: + if not path.is_dir(): + _fail(f"{path} is not a directory or Office file") + unpacked_dir = path + + match family: + case "docx": + validators = [ + DOCXSchemaValidator(unpacked_dir, original_file, verbose=args.verbose), + ] + if args.author is not None: + validators.append( + RedliningValidator(unpacked_dir, original_file, verbose=args.verbose) + ) + elif original_file and _has_tracked_changes(unpacked_dir): + print( + "Note: this document has tracked changes; they were not " + "checked against the original (pass --author to check)." + ) + case "pptx": + validators = [ + PPTXSchemaValidator(unpacked_dir, original_file, verbose=args.verbose), + ] + case "xlsx": + exts = ", ".join(k for k, v in sorted(OOXML_FAMILY.items()) if v == "xlsx") + print( + f"No XSD schema validation is performed for xlsx-family files ({exts}). " + "For formula-error checking, use scripts/recalc.py instead." + ) + sys.exit(0) + case _: + print(f"Error: Validation not supported for file type {family}") + sys.exit(1) + + if args.auto_repair: + total_repairs = sum(v.repair() for v in validators) + if total_repairs: + print(f"Auto-repaired {total_repairs} issue(s)") + if packed_file is not None: + rezip(unpacked_dir, packed_file) + print(f"Wrote repaired file to {packed_file}") + + success = all([v.validate() for v in validators]) + + if temp_dir_ctx is not None: + temp_dir_ctx.cleanup() + + if success: + print("All validations PASSED!") + + sys.exit(0 if success else 1) + + +if __name__ == "__main__": + main() diff --git a/.github/skills/anthropic-xlsx/scripts/office/validators/__init__.py b/.github/skills/anthropic-xlsx/scripts/office/validators/__init__.py new file mode 100644 index 00000000..db092ece --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/validators/__init__.py @@ -0,0 +1,15 @@ +""" +Validation modules for Word document processing. +""" + +from .base import BaseSchemaValidator +from .docx import DOCXSchemaValidator +from .pptx import PPTXSchemaValidator +from .redlining import RedliningValidator + +__all__ = [ + "BaseSchemaValidator", + "DOCXSchemaValidator", + "PPTXSchemaValidator", + "RedliningValidator", +] diff --git a/.github/skills/anthropic-xlsx/scripts/office/validators/base.py b/.github/skills/anthropic-xlsx/scripts/office/validators/base.py new file mode 100644 index 00000000..33fc97bb --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/validators/base.py @@ -0,0 +1,875 @@ +""" +Base validator with common validation logic for document files. +""" + +import re +from pathlib import Path + +import defusedxml.minidom +from functools import lru_cache + +import lxml.etree + +from helpers import safe_extract + + +@lru_cache(maxsize=None) +def _load_schema(schema_path: str): + with open(schema_path, "rb") as xsd_file: + xsd_doc = lxml.etree.parse( + xsd_file, parser=lxml.etree.XMLParser(), base_url=schema_path + ) + return lxml.etree.XMLSchema(xsd_doc) + +class BaseSchemaValidator: + + IGNORED_VALIDATION_ERRORS = [ + "hyphenationZone", + "purl.org/dc/terms", + ] + + UNIQUE_ID_REQUIREMENTS = { + "comment": ("id", "file"), + "commentrangestart": ("id", "file"), + "commentrangeend": ("id", "file"), + "bookmarkstart": ("id", "file"), + "bookmarkend": ("id", "file"), + "sldid": ("id", "file"), + "sldmasterid": ("id", "global"), + "sldlayoutid": ("id", "global"), + "cm": ("authorid", "file"), + "sheet": ("sheetid", "file"), + "definedname": ("id", "file"), + "cxnsp": ("id", "file"), + "sp": ("id", "file"), + "pic": ("id", "file"), + "grpsp": ("id", "file"), + } + + EXCLUDED_ID_CONTAINERS = { + "sectionlst", + } + + ELEMENT_RELATIONSHIP_TYPES = {} + + SCHEMA_MAPPINGS = { + "word": "ISO-IEC29500-4_2016/wml.xsd", + "ppt": "ISO-IEC29500-4_2016/pml.xsd", + "xl": "ISO-IEC29500-4_2016/sml.xsd", + "[Content_Types].xml": "ecma/fouth-edition/opc-contentTypes.xsd", + "app.xml": "ISO-IEC29500-4_2016/shared-documentPropertiesExtended.xsd", + "core.xml": "ecma/fouth-edition/opc-coreProperties.xsd", + "custom.xml": "ISO-IEC29500-4_2016/shared-documentPropertiesCustom.xsd", + ".rels": "ecma/fouth-edition/opc-relationships.xsd", + "people.xml": "microsoft/wml-2012.xsd", + "commentsIds.xml": "microsoft/wml-cid-2016.xsd", + "commentsExtensible.xml": "microsoft/wml-cex-2018.xsd", + "commentsExtended.xml": "microsoft/wml-2012.xsd", + "chart": "ISO-IEC29500-4_2016/dml-chart.xsd", + "theme": "ISO-IEC29500-4_2016/dml-main.xsd", + "drawing": "ISO-IEC29500-4_2016/dml-main.xsd", + } + + MC_NAMESPACE = "http://schemas.openxmlformats.org/markup-compatibility/2006" + XML_NAMESPACE = "http://www.w3.org/XML/1998/namespace" + + PACKAGE_RELATIONSHIPS_NAMESPACE = ( + "http://schemas.openxmlformats.org/package/2006/relationships" + ) + OFFICE_RELATIONSHIPS_NAMESPACE = ( + "http://schemas.openxmlformats.org/officeDocument/2006/relationships" + ) + CONTENT_TYPES_NAMESPACE = ( + "http://schemas.openxmlformats.org/package/2006/content-types" + ) + + MAIN_CONTENT_FOLDERS = {"word", "ppt", "xl"} + + OOXML_NAMESPACES = { + "http://schemas.openxmlformats.org/officeDocument/2006/math", + "http://schemas.openxmlformats.org/officeDocument/2006/relationships", + "http://schemas.openxmlformats.org/schemaLibrary/2006/main", + "http://schemas.openxmlformats.org/drawingml/2006/main", + "http://schemas.openxmlformats.org/drawingml/2006/chart", + "http://schemas.openxmlformats.org/drawingml/2006/chartDrawing", + "http://schemas.openxmlformats.org/drawingml/2006/diagram", + "http://schemas.openxmlformats.org/drawingml/2006/picture", + "http://schemas.openxmlformats.org/drawingml/2006/spreadsheetDrawing", + "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing", + "http://schemas.openxmlformats.org/wordprocessingml/2006/main", + "http://schemas.openxmlformats.org/presentationml/2006/main", + "http://schemas.openxmlformats.org/spreadsheetml/2006/main", + "http://schemas.openxmlformats.org/officeDocument/2006/sharedTypes", + "http://www.w3.org/XML/1998/namespace", + } + + def __init__(self, unpacked_dir, original_file=None, verbose=False): + self.unpacked_dir = Path(unpacked_dir).resolve() + self.original_file = Path(original_file) if original_file else None + self.verbose = verbose + + self.schemas_dir = Path(__file__).parent.parent / "schemas" + + patterns = ["*.xml", "*.rels"] + self.xml_files = [ + f for pattern in patterns for f in self.unpacked_dir.rglob(pattern) + ] + + if not self.xml_files: + print(f"Warning: No XML files found in {self.unpacked_dir}") + + def validate(self): + raise NotImplementedError("Subclasses must implement the validate method") + + def repair(self) -> int: + return self.repair_whitespace_preservation() + + def repair_whitespace_preservation(self) -> int: + repairs = 0 + + for xml_file in self.xml_files: + try: + content = xml_file.read_text(encoding="utf-8") + dom = defusedxml.minidom.parseString(content) + pending = [] + + for elem in dom.getElementsByTagName("*"): + local_name = elem.tagName.rsplit(":", 1)[-1] + if local_name in ("t", "delText", "instrText", "delInstrText"): + text = "".join( + child.data + for child in elem.childNodes + if child.nodeType in (child.TEXT_NODE, child.CDATA_SECTION_NODE) + ) + ws = (" ", "\t", "\n", "\r") + if text and (text.startswith(ws) or text.endswith(ws)): + if elem.getAttribute("xml:space") != "preserve": + elem.setAttribute("xml:space", "preserve") + text_preview = repr(text[:30]) + "..." if len(text) > 30 else repr(text) + pending.append(f" Repaired: {xml_file.name}: Added xml:space='preserve' to {elem.tagName}: {text_preview}") + + if pending: + xml_file.write_bytes(dom.toxml(encoding="UTF-8")) + for message in pending: + print(message) + repairs += len(pending) + + except Exception: + pass + + return repairs + + def validate_xml(self): + errors = [] + + for xml_file in self.xml_files: + try: + lxml.etree.parse(str(xml_file)) + except lxml.etree.XMLSyntaxError as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {e.lineno}: {e.msg}" + ) + except Exception as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Unexpected error: {str(e)}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} XML violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All XML files are well-formed") + return True + + def validate_namespaces(self): + errors = [] + + for xml_file in self.xml_files: + try: + root = lxml.etree.parse(str(xml_file)).getroot() + declared = set(root.nsmap.keys()) - {None} + + for attr_val in [ + v for k, v in root.attrib.items() if k.endswith("Ignorable") + ]: + undeclared = set(attr_val.split()) - declared + errors.extend( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Namespace '{ns}' in Ignorable but not declared" + for ns in undeclared + ) + except lxml.etree.XMLSyntaxError: + continue + + if errors: + print(f"FAILED - {len(errors)} namespace issues:") + for error in errors: + print(error) + return False + if self.verbose: + print("PASSED - All namespace prefixes properly declared") + return True + + def validate_unique_ids(self): + errors = [] + global_ids = {} + + for xml_file in self.xml_files: + try: + root = lxml.etree.parse(str(xml_file)).getroot() + file_ids = {} + + mc_elements = root.xpath( + ".//mc:AlternateContent", namespaces={"mc": self.MC_NAMESPACE} + ) + for elem in mc_elements: + elem.getparent().remove(elem) + + for elem in root.iter(): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + tag = ( + elem.tag.split("}")[-1].lower() + if "}" in elem.tag + else elem.tag.lower() + ) + + if tag in self.UNIQUE_ID_REQUIREMENTS: + in_excluded_container = any( + ancestor.tag.split("}")[-1].lower() in self.EXCLUDED_ID_CONTAINERS + for ancestor in elem.iterancestors() + ) + if in_excluded_container: + continue + + attr_name, scope = self.UNIQUE_ID_REQUIREMENTS[tag] + + id_value = None + for attr, value in elem.attrib.items(): + attr_local = ( + attr.split("}")[-1].lower() + if "}" in attr + else attr.lower() + ) + if attr_local == attr_name: + id_value = value + break + + if id_value is not None: + if scope == "global": + if id_value in global_ids: + prev_file, prev_line, prev_tag = global_ids[ + id_value + ] + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: Global ID '{id_value}' in <{tag}> " + f"already used in {prev_file} at line {prev_line} in <{prev_tag}>" + ) + else: + global_ids[id_value] = ( + xml_file.relative_to(self.unpacked_dir), + elem.sourceline, + tag, + ) + elif scope == "file": + key = (tag, attr_name) + if key not in file_ids: + file_ids[key] = {} + + if id_value in file_ids[key]: + prev_line = file_ids[key][id_value] + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: Duplicate {attr_name}='{id_value}' in <{tag}> " + f"(first occurrence at line {prev_line})" + ) + else: + file_ids[key][id_value] = elem.sourceline + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} ID uniqueness violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All required IDs are unique") + return True + + def validate_file_references(self): + errors = [] + + rels_files = list(self.unpacked_dir.rglob("*.rels")) + + if not rels_files: + if self.verbose: + print("PASSED - No .rels files found") + return True + + all_files = [] + for file_path in self.unpacked_dir.rglob("*"): + if ( + file_path.is_file() + and file_path.name != "[Content_Types].xml" + and not file_path.name.endswith(".rels") + ): + all_files.append(file_path.resolve()) + + all_referenced_files = set() + + if self.verbose: + print( + f"Found {len(rels_files)} .rels files and {len(all_files)} target files" + ) + + for rels_file in rels_files: + try: + rels_root = lxml.etree.parse(str(rels_file)).getroot() + + rels_dir = rels_file.parent + + referenced_files = set() + broken_refs = [] + + for rel in rels_root.findall( + ".//ns:Relationship", + namespaces={"ns": self.PACKAGE_RELATIONSHIPS_NAMESPACE}, + ): + target = rel.get("Target") + if rel.get("TargetMode") == "External": + continue + if target and not target.startswith( + ("http", "mailto:") + ): + if target.startswith("/"): + target_path = self.unpacked_dir / target.lstrip("/") + elif rels_file.name == ".rels": + target_path = self.unpacked_dir / target + else: + base_dir = rels_dir.parent + target_path = base_dir / target + + try: + target_path = target_path.resolve() + if target_path.exists() and target_path.is_file(): + referenced_files.add(target_path) + all_referenced_files.add(target_path) + else: + broken_refs.append((target, rel.sourceline)) + except (OSError, ValueError): + broken_refs.append((target, rel.sourceline)) + + if broken_refs: + rel_path = rels_file.relative_to(self.unpacked_dir) + for broken_ref, line_num in broken_refs: + errors.append( + f" {rel_path}: Line {line_num}: Broken reference to {broken_ref}" + ) + + except Exception as e: + rel_path = rels_file.relative_to(self.unpacked_dir) + errors.append(f" Error parsing {rel_path}: {e}") + + unreferenced_files = set(all_files) - all_referenced_files + + if unreferenced_files: + for unref_file in sorted(unreferenced_files): + unref_rel_path = unref_file.relative_to(self.unpacked_dir) + errors.append(f" Unreferenced file: {unref_rel_path}") + + if errors: + print(f"FAILED - Found {len(errors)} relationship validation errors:") + for error in errors: + print(error) + print( + "CRITICAL: These errors will cause the document to appear corrupt. " + + "Broken references MUST be fixed, " + + "and unreferenced files MUST be referenced or removed." + ) + return False + else: + if self.verbose: + print( + "PASSED - All references are valid and all files are properly referenced" + ) + return True + + def validate_all_relationship_ids(self): + import lxml.etree + + errors = [] + + for xml_file in self.xml_files: + if xml_file.suffix == ".rels": + continue + + rels_dir = xml_file.parent / "_rels" + rels_file = rels_dir / f"{xml_file.name}.rels" + + if not rels_file.exists(): + continue + + try: + rels_root = lxml.etree.parse(str(rels_file)).getroot() + rid_to_type = {} + + for rel in rels_root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ): + rid = rel.get("Id") + rel_type = rel.get("Type", "") + if rid: + if rid in rid_to_type: + rels_rel_path = rels_file.relative_to(self.unpacked_dir) + errors.append( + f" {rels_rel_path}: Line {rel.sourceline}: " + f"Duplicate relationship ID '{rid}' (IDs must be unique)" + ) + type_name = ( + rel_type.split("/")[-1] if "/" in rel_type else rel_type + ) + rid_to_type[rid] = type_name + + xml_root = lxml.etree.parse(str(xml_file)).getroot() + + r_ns = self.OFFICE_RELATIONSHIPS_NAMESPACE + rid_attrs_to_check = ["id", "embed", "link"] + for elem in xml_root.iter(): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + for attr_name in rid_attrs_to_check: + rid_attr = elem.get(f"{{{r_ns}}}{attr_name}") + if not rid_attr: + continue + xml_rel_path = xml_file.relative_to(self.unpacked_dir) + elem_name = ( + elem.tag.split("}")[-1] if "}" in elem.tag else elem.tag + ) + + if rid_attr not in rid_to_type: + errors.append( + f" {xml_rel_path}: Line {elem.sourceline}: " + f"<{elem_name}> r:{attr_name} references non-existent relationship '{rid_attr}' " + f"(valid IDs: {', '.join(sorted(rid_to_type.keys())[:5])}{'...' if len(rid_to_type) > 5 else ''})" + ) + elif attr_name == "id" and self.ELEMENT_RELATIONSHIP_TYPES: + expected_type = self._get_expected_relationship_type( + elem_name + ) + if expected_type: + actual_type = rid_to_type[rid_attr] + if expected_type not in actual_type.lower(): + errors.append( + f" {xml_rel_path}: Line {elem.sourceline}: " + f"<{elem_name}> references '{rid_attr}' which points to '{actual_type}' " + f"but should point to a '{expected_type}' relationship" + ) + + except Exception as e: + xml_rel_path = xml_file.relative_to(self.unpacked_dir) + errors.append(f" Error processing {xml_rel_path}: {e}") + + if errors: + print(f"FAILED - Found {len(errors)} relationship ID reference errors:") + for error in errors: + print(error) + print("\nThese ID mismatches will cause the document to appear corrupt!") + return False + else: + if self.verbose: + print("PASSED - All relationship ID references are valid") + return True + + def _get_expected_relationship_type(self, element_name): + elem_lower = element_name.lower() + + if elem_lower in self.ELEMENT_RELATIONSHIP_TYPES: + return self.ELEMENT_RELATIONSHIP_TYPES[elem_lower] + + if elem_lower.endswith("id") and len(elem_lower) > 2: + prefix = elem_lower[:-2] + if prefix.endswith("master"): + return prefix.lower() + elif prefix.endswith("layout"): + return prefix.lower() + else: + if prefix == "sld": + return "slide" + return prefix.lower() + + if elem_lower.endswith("reference") and len(elem_lower) > 9: + prefix = elem_lower[:-9] + return prefix.lower() + + return None + + def validate_content_types(self): + errors = [] + + content_types_file = self.unpacked_dir / "[Content_Types].xml" + if not content_types_file.exists(): + print("FAILED - [Content_Types].xml file not found") + return False + + try: + root = lxml.etree.parse(str(content_types_file)).getroot() + declared_parts = set() + declared_extensions = set() + + for override in root.findall( + f".//{{{self.CONTENT_TYPES_NAMESPACE}}}Override" + ): + part_name = override.get("PartName") + if part_name is not None: + declared_parts.add(part_name.lstrip("/")) + + for default in root.findall( + f".//{{{self.CONTENT_TYPES_NAMESPACE}}}Default" + ): + extension = default.get("Extension") + if extension is not None: + declared_extensions.add(extension.lower()) + + declarable_roots = { + "sld", + "sldLayout", + "sldMaster", + "presentation", + "document", + "workbook", + "worksheet", + "theme", + } + + media_extensions = { + "png": "image/png", + "jpg": "image/jpeg", + "jpeg": "image/jpeg", + "gif": "image/gif", + "bmp": "image/bmp", + "tiff": "image/tiff", + "wmf": "image/x-wmf", + "emf": "image/x-emf", + } + + all_files = list(self.unpacked_dir.rglob("*")) + all_files = [f for f in all_files if f.is_file()] + + for xml_file in self.xml_files: + path_str = str(xml_file.relative_to(self.unpacked_dir)).replace( + "\\", "/" + ) + + if any( + skip in path_str + for skip in [".rels", "[Content_Types]", "docProps/", "_rels/"] + ): + continue + + try: + root_tag = lxml.etree.parse(str(xml_file)).getroot().tag + root_name = root_tag.split("}")[-1] if "}" in root_tag else root_tag + + if root_name in declarable_roots and path_str not in declared_parts: + errors.append( + f" {path_str}: File with <{root_name}> root not declared in [Content_Types].xml" + ) + + except Exception: + continue + + for file_path in all_files: + if file_path.suffix.lower() in {".xml", ".rels"}: + continue + if file_path.name == "[Content_Types].xml": + continue + if "_rels" in file_path.parts or "docProps" in file_path.parts: + continue + + extension = file_path.suffix.lstrip(".").lower() + if extension and extension not in declared_extensions: + if extension in media_extensions: + relative_path = file_path.relative_to(self.unpacked_dir) + errors.append( + f' {relative_path}: File with extension \'{extension}\' not declared in [Content_Types].xml - should add: ' + ) + + except Exception as e: + errors.append(f" Error parsing [Content_Types].xml: {e}") + + if errors: + print(f"FAILED - Found {len(errors)} content type declaration errors:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print( + "PASSED - All content files are properly declared in [Content_Types].xml" + ) + return True + + def validate_file_against_xsd(self, xml_file, verbose=False): + xml_file = Path(xml_file).resolve() + unpacked_dir = self.unpacked_dir.resolve() + + is_valid, current_errors = self._validate_single_file_xsd( + xml_file, unpacked_dir + ) + + if is_valid is None: + return None, set() + elif is_valid: + return True, set() + + original_errors = self._get_original_file_errors(xml_file) + + assert current_errors is not None + new_errors = current_errors - original_errors + + new_errors = { + e for e in new_errors + if not any(pattern in e for pattern in self.IGNORED_VALIDATION_ERRORS) + } + + if new_errors: + if verbose: + relative_path = xml_file.relative_to(unpacked_dir) + print(f"FAILED - {relative_path}: {len(new_errors)} new error(s)") + for error in list(new_errors)[:3]: + truncated = error[:250] + "..." if len(error) > 250 else error + print(f" - {truncated}") + return False, new_errors + else: + if verbose: + print( + f"PASSED - No new errors (original had {len(current_errors)} errors)" + ) + return True, set() + + def validate_against_xsd(self): + new_errors = [] + original_error_count = 0 + valid_count = 0 + skipped_count = 0 + + for xml_file in self.xml_files: + relative_path = str(xml_file.relative_to(self.unpacked_dir)) + is_valid, new_file_errors = self.validate_file_against_xsd( + xml_file, verbose=False + ) + + if is_valid is None: + skipped_count += 1 + continue + elif is_valid and not new_file_errors: + valid_count += 1 + continue + elif is_valid: + original_error_count += 1 + valid_count += 1 + continue + + new_errors.append(f" {relative_path}: {len(new_file_errors)} new error(s)") + for error in list(new_file_errors)[:3]: + new_errors.append( + f" - {error[:250]}..." if len(error) > 250 else f" - {error}" + ) + + if self.verbose: + print(f"Validated {len(self.xml_files)} files:") + print(f" - Valid: {valid_count}") + print(f" - Skipped (no schema): {skipped_count}") + if original_error_count: + print(f" - With original errors (ignored): {original_error_count}") + print( + f" - With NEW errors: {len(new_errors) > 0 and len([e for e in new_errors if not e.startswith(' ')]) or 0}" + ) + + if new_errors: + print("\nFAILED - Found NEW validation errors:") + for error in new_errors: + print(error) + return False + else: + if self.verbose: + print("\nPASSED - No new XSD validation errors introduced") + return True + + def _get_schema_path(self, xml_file): + if xml_file.name in self.SCHEMA_MAPPINGS: + return self.schemas_dir / self.SCHEMA_MAPPINGS[xml_file.name] + + if xml_file.suffix == ".rels": + return self.schemas_dir / self.SCHEMA_MAPPINGS[".rels"] + + if "charts/" in str(xml_file) and xml_file.name.startswith("chart"): + return self.schemas_dir / self.SCHEMA_MAPPINGS["chart"] + + if "theme/" in str(xml_file) and xml_file.name.startswith("theme"): + return self.schemas_dir / self.SCHEMA_MAPPINGS["theme"] + + if xml_file.parent.name in self.MAIN_CONTENT_FOLDERS: + return self.schemas_dir / self.SCHEMA_MAPPINGS[xml_file.parent.name] + + return None + + def _clean_ignorable_namespaces(self, xml_doc): + xml_string = lxml.etree.tostring(xml_doc, encoding="unicode") + xml_copy = lxml.etree.fromstring(xml_string) + + for elem in xml_copy.iter(): + attrs_to_remove = [] + + for attr in elem.attrib: + if "{" in attr: + ns = attr.split("}")[0][1:] + if ns not in self.OOXML_NAMESPACES: + attrs_to_remove.append(attr) + + for attr in attrs_to_remove: + del elem.attrib[attr] + + self._remove_ignorable_elements(xml_copy) + + return lxml.etree.ElementTree(xml_copy) + + def _remove_ignorable_elements(self, root): + elements_to_remove = [] + + for elem in list(root): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + + tag_str = str(elem.tag) + if tag_str.startswith("{"): + ns = tag_str.split("}")[0][1:] + if ns not in self.OOXML_NAMESPACES: + elements_to_remove.append(elem) + continue + + self._remove_ignorable_elements(elem) + + for elem in elements_to_remove: + root.remove(elem) + + def _preprocess_for_mc_ignorable(self, xml_doc): + root = xml_doc.getroot() + + if f"{{{self.MC_NAMESPACE}}}Ignorable" in root.attrib: + del root.attrib[f"{{{self.MC_NAMESPACE}}}Ignorable"] + + return xml_doc + + def _preprocess_for_schema(self, xml_doc, relative_path): + return xml_doc + + def _validate_single_file_xsd(self, xml_file, base_path, schema_path=None): + schema_path = schema_path or self._get_schema_path(xml_file) + if not schema_path: + return None, None + + try: + schema = _load_schema(str(schema_path)) + + with open(xml_file, "r") as f: + xml_doc = lxml.etree.parse(f) + + xml_doc, _ = self._remove_template_tags_from_text_nodes(xml_doc) + xml_doc = self._preprocess_for_mc_ignorable(xml_doc) + + relative_path = xml_file.relative_to(base_path) + if ( + relative_path.parts + and relative_path.parts[0] in self.MAIN_CONTENT_FOLDERS + ): + xml_doc = self._clean_ignorable_namespaces(xml_doc) + + xml_doc = self._preprocess_for_schema(xml_doc, relative_path) + + if schema.validate(xml_doc): + return True, set() + else: + errors = set() + for error in schema.error_log: + errors.add(error.message) + return False, errors + + except Exception as e: + return False, {str(e)} + + def _get_original_file_errors(self, xml_file, schema_path=None): + if self.original_file is None: + return set() + + import tempfile + import zipfile + + xml_file = Path(xml_file).resolve() + unpacked_dir = self.unpacked_dir.resolve() + relative_path = xml_file.relative_to(unpacked_dir) + + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + + try: + with zipfile.ZipFile(self.original_file, "r") as zip_ref: + safe_extract(zip_ref, temp_path) + except (zipfile.BadZipFile, ValueError, OSError): + return set() + + original_xml_file = temp_path / relative_path + + if not original_xml_file.exists(): + return set() + + is_valid, errors = self._validate_single_file_xsd( + original_xml_file, temp_path, schema_path=schema_path + ) + return errors if errors else set() + + def _remove_template_tags_from_text_nodes(self, xml_doc): + warnings = [] + template_pattern = re.compile(r"\{\{[^}]*\}\}") + + xml_string = lxml.etree.tostring(xml_doc, encoding="unicode") + xml_copy = lxml.etree.fromstring(xml_string) + + def process_text_content(text, content_type): + if not text: + return text + matches = list(template_pattern.finditer(text)) + if matches: + for match in matches: + warnings.append( + f"Found template tag in {content_type}: {match.group()}" + ) + return template_pattern.sub("", text) + return text + + for elem in xml_copy.iter(): + if not hasattr(elem, "tag") or callable(elem.tag): + continue + tag_str = str(elem.tag) + if tag_str.endswith("}t") or tag_str == "t": + continue + + elem.text = process_text_content(elem.text, "text content") + elem.tail = process_text_content(elem.tail, "tail content") + + return lxml.etree.ElementTree(xml_copy), warnings + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-xlsx/scripts/office/validators/docx.py b/.github/skills/anthropic-xlsx/scripts/office/validators/docx.py new file mode 100644 index 00000000..b1814994 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/validators/docx.py @@ -0,0 +1,466 @@ +""" +Validator for Word document XML files against XSD schemas. +""" + +import random +import re +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.minidom +import lxml.etree + +from helpers import safe_extract + +from .base import BaseSchemaValidator + + +class DOCXSchemaValidator(BaseSchemaValidator): + + WORD_2006_NAMESPACE = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + W14_NAMESPACE = "http://schemas.microsoft.com/office/word/2010/wordml" + W16CID_NAMESPACE = "http://schemas.microsoft.com/office/word/2016/wordml/cid" + + ELEMENT_RELATIONSHIP_TYPES = {} + + def validate(self): + if not self.validate_xml(): + return False + + all_valid = True + if not self.validate_namespaces(): + all_valid = False + + if not self.validate_unique_ids(): + all_valid = False + + if not self.validate_file_references(): + all_valid = False + + if not self.validate_content_types(): + all_valid = False + + if not self.validate_against_xsd(): + all_valid = False + + if not self.validate_whitespace_preservation(): + all_valid = False + + if not self.validate_deletions(): + all_valid = False + + if not self.validate_insertions(): + all_valid = False + + if not self.validate_all_relationship_ids(): + all_valid = False + + if not self.validate_id_constraints(): + all_valid = False + + if not self.validate_comment_markers(): + all_valid = False + + self.compare_paragraph_counts() + + return all_valid + + def validate_whitespace_preservation(self): + errors = [] + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + + for elem in root.iter(f"{{{self.WORD_2006_NAMESPACE}}}t"): + if elem.text: + text = elem.text + if re.search(r"^[ \t\n\r]", text) or re.search( + r"[ \t\n\r]$", text + ): + xml_space_attr = f"{{{self.XML_NAMESPACE}}}space" + if ( + xml_space_attr not in elem.attrib + or elem.attrib[xml_space_attr] != "preserve" + ): + text_preview = ( + repr(text)[:50] + "..." + if len(repr(text)) > 50 + else repr(text) + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: w:t element with whitespace missing xml:space='preserve': {text_preview}" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} whitespace preservation violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All whitespace is properly preserved") + return True + + def validate_deletions(self): + errors = [] + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + namespaces = {"w": self.WORD_2006_NAMESPACE} + + for t_elem in root.xpath(".//w:del//w:t", namespaces=namespaces): + if t_elem.text: + text_preview = ( + repr(t_elem.text)[:50] + "..." + if len(repr(t_elem.text)) > 50 + else repr(t_elem.text) + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {t_elem.sourceline}: found within : {text_preview}" + ) + + for instr_elem in root.xpath( + ".//w:del//w:instrText", namespaces=namespaces + ): + text_preview = ( + repr(instr_elem.text or "")[:50] + "..." + if len(repr(instr_elem.text or "")) > 50 + else repr(instr_elem.text or "") + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {instr_elem.sourceline}: found within (use ): {text_preview}" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} deletion validation violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - No w:t elements found within w:del elements") + return True + + def count_paragraphs_in_unpacked(self): + count = 0 + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + paragraphs = root.findall(f".//{{{self.WORD_2006_NAMESPACE}}}p") + count = len(paragraphs) + except Exception as e: + print(f"Error counting paragraphs in unpacked document: {e}") + + return count + + def count_paragraphs_in_original(self): + original = self.original_file + if original is None: + return 0 + + count = 0 + + try: + with tempfile.TemporaryDirectory() as temp_dir: + with zipfile.ZipFile(original, "r") as zip_ref: + safe_extract(zip_ref, Path(temp_dir)) + + doc_xml_path = temp_dir + "/word/document.xml" + root = lxml.etree.parse(doc_xml_path).getroot() + + paragraphs = root.findall(f".//{{{self.WORD_2006_NAMESPACE}}}p") + count = len(paragraphs) + + except Exception as e: + print(f"Error counting paragraphs in original document: {e}") + + return count + + def validate_insertions(self): + errors = [] + + for xml_file in self.xml_files: + if xml_file.name != "document.xml": + continue + + try: + root = lxml.etree.parse(str(xml_file)).getroot() + namespaces = {"w": self.WORD_2006_NAMESPACE} + + invalid_elements = root.xpath( + ".//w:ins//w:delText[not(ancestor::w:del)]", namespaces=namespaces + ) + + for elem in invalid_elements: + text_preview = ( + repr(elem.text or "")[:50] + "..." + if len(repr(elem.text or "")) > 50 + else repr(elem.text or "") + ) + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: within : {text_preview}" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} insertion validation violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - No w:delText elements within w:ins elements") + return True + + def compare_paragraph_counts(self): + new_count = self.count_paragraphs_in_unpacked() + if self.original_file is None: + print(f"\nParagraphs: {new_count}") + return + + original_count = self.count_paragraphs_in_original() + diff = new_count - original_count + diff_str = f"+{diff}" if diff > 0 else str(diff) + print(f"\nParagraphs: {original_count} → {new_count} ({diff_str})") + + def _parse_id_value(self, val: str, base: int = 16) -> int: + return int(val, base) + + def validate_id_constraints(self): + errors = [] + para_id_attr = f"{{{self.W14_NAMESPACE}}}paraId" + durable_id_attr = f"{{{self.W16CID_NAMESPACE}}}durableId" + + for xml_file in self.xml_files: + try: + for elem in lxml.etree.parse(str(xml_file)).iter(): + if val := elem.get(para_id_attr): + try: + if self._parse_id_value(val, base=16) >= 0x80000000: + errors.append( + f" {xml_file.name}:{elem.sourceline}: paraId={val} >= 0x80000000" + ) + except ValueError: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"paraId={val} is not valid hex" + ) + + if val := elem.get(durable_id_attr): + if xml_file.name == "numbering.xml": + try: + if self._parse_id_value(val, base=10) >= 0x7FFFFFFF: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} >= 0x7FFFFFFF" + ) + except ValueError: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} must be decimal in numbering.xml" + ) + else: + try: + if self._parse_id_value(val, base=16) >= 0x7FFFFFFF: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} >= 0x7FFFFFFF" + ) + except ValueError: + errors.append( + f" {xml_file.name}:{elem.sourceline}: " + f"durableId={val} is not valid hex" + ) + except lxml.etree.XMLSyntaxError: + continue + + if errors: + print(f"FAILED - {len(errors)} ID constraint violations:") + for e in errors: + print(e) + elif self.verbose: + print("PASSED - All paraId/durableId values within constraints") + return not errors + + def validate_comment_markers(self): + errors = [] + + document_xml = None + comments_xml = None + for xml_file in self.xml_files: + if xml_file.name == "document.xml" and "word" in str(xml_file): + document_xml = xml_file + elif xml_file.name == "comments.xml": + comments_xml = xml_file + + if not document_xml: + if self.verbose: + print("PASSED - No document.xml found (skipping comment validation)") + return True + + try: + doc_root = lxml.etree.parse(str(document_xml)).getroot() + namespaces = {"w": self.WORD_2006_NAMESPACE} + + range_starts = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in doc_root.xpath( + ".//w:commentRangeStart", namespaces=namespaces + ) + } + range_ends = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in doc_root.xpath( + ".//w:commentRangeEnd", namespaces=namespaces + ) + } + references = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in doc_root.xpath( + ".//w:commentReference", namespaces=namespaces + ) + } + + orphaned_ends = range_ends - range_starts + for comment_id in sorted( + orphaned_ends, key=lambda x: int(x) if x and x.isdigit() else 0 + ): + errors.append( + f' document.xml: commentRangeEnd id="{comment_id}" has no matching commentRangeStart' + ) + + orphaned_starts = range_starts - range_ends + for comment_id in sorted( + orphaned_starts, key=lambda x: int(x) if x and x.isdigit() else 0 + ): + errors.append( + f' document.xml: commentRangeStart id="{comment_id}" has no matching commentRangeEnd' + ) + + comment_ids = set() + if comments_xml and comments_xml.exists(): + comments_root = lxml.etree.parse(str(comments_xml)).getroot() + comment_ids = { + elem.get(f"{{{self.WORD_2006_NAMESPACE}}}id") + for elem in comments_root.xpath( + ".//w:comment", namespaces=namespaces + ) + } + + marker_ids = range_starts | range_ends | references + invalid_refs = marker_ids - comment_ids + for comment_id in sorted( + invalid_refs, key=lambda x: int(x) if x and x.isdigit() else 0 + ): + if comment_id: + errors.append( + f' document.xml: marker id="{comment_id}" references non-existent comment' + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append(f" Error parsing XML: {e}") + + if errors: + print(f"FAILED - {len(errors)} comment marker violations:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All comment markers properly paired") + return True + + def repair(self) -> int: + repairs = super().repair() + repairs += self.repair_durableId() + return repairs + + def repair_durableId(self) -> int: + DURABLE_ID_ATTRS = ("w16cid:durableId", "w16cex:durableId") + repairs = 0 + renames: dict = {} + + for xml_file in self.xml_files: + try: + content = xml_file.read_text(encoding="utf-8") + dom = defusedxml.minidom.parseString(content) + is_numbering = xml_file.name == "numbering.xml" + base = 10 if is_numbering else 16 + pending = [] + seen_in_file = set() + modified = False + + for elem in dom.getElementsByTagName("*"): + for attr_name in DURABLE_ID_ATTRS: + if not elem.hasAttribute(attr_name): + continue + + durable_id = elem.getAttribute(attr_name) + try: + key = self._parse_id_value(durable_id, base=base) + needs_repair = key >= 0x7FFFFFFF + except ValueError: + key = durable_id + needs_repair = True + + if needs_repair: + if key in seen_in_file: + value = random.randint(1, 0x7FFFFFFE) + else: + seen_in_file.add(key) + if key not in renames: + renames[key] = random.randint(1, 0x7FFFFFFE) + value = renames[key] + new_id = str(value) if is_numbering else f"{value:08X}" + + elem.setAttribute(attr_name, new_id) + pending.append( + f" Repaired: {xml_file.name}: durableId {durable_id} → {new_id}" + ) + modified = True + + if modified: + xml_file.write_bytes(dom.toxml(encoding="UTF-8")) + for message in pending: + print(message) + repairs += len(pending) + + except Exception: + pass + + return repairs + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-xlsx/scripts/office/validators/pptx.py b/.github/skills/anthropic-xlsx/scripts/office/validators/pptx.py new file mode 100644 index 00000000..318f0e61 --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/validators/pptx.py @@ -0,0 +1,441 @@ +""" +Validator for PowerPoint presentation XML files against XSD schemas. +""" + +import re +from pathlib import Path + +from helpers import opc_target, rels_source_part, safe_extract + +from .base import BaseSchemaValidator + + +class PPTXSchemaValidator(BaseSchemaValidator): + + PRESENTATIONML_NAMESPACE = ( + "http://schemas.openxmlformats.org/presentationml/2006/main" + ) + + ELEMENT_RELATIONSHIP_TYPES = { + "sldid": "slide", + "sldmasterid": "slidemaster", + "notesmasterid": "notesmaster", + "sldlayoutid": "slidelayout", + "themeid": "theme", + "tablestyleid": "tablestyles", + } + + def validate(self): + if not self.validate_xml(): + return False + + all_valid = True + if not self.validate_namespaces(): + all_valid = False + + if not self.validate_unique_ids(): + all_valid = False + + if not self.validate_uuid_ids(): + all_valid = False + + if not self.validate_file_references(): + all_valid = False + + if not self.validate_slide_layout_ids(): + all_valid = False + + if not self.validate_content_types(): + all_valid = False + + if not self.validate_against_xsd(): + all_valid = False + + if not self.validate_notes_slide_references(): + all_valid = False + + if not self.validate_all_relationship_ids(): + all_valid = False + + if not self.validate_no_duplicate_slide_layouts(): + all_valid = False + + if not self.validate_master_theme_uniqueness(): + all_valid = False + + if not self.validate_charts(): + all_valid = False + + if not self.validate_slides(): + all_valid = False + + return all_valid + + def _package_map(self) -> dict: + wanted = [] + wanted += list(self.unpacked_dir.glob("[[]Content_Types[]].xml")) + wanted += list(self.unpacked_dir.glob("ppt/presentation.xml")) + wanted += list(self.unpacked_dir.glob("ppt/theme/*.xml")) + wanted += list(self.unpacked_dir.glob("ppt/theme/_rels/*.rels")) + wanted += list(self.unpacked_dir.glob("ppt/charts/chart*.xml")) + for group in ("slideMasters", "notesMasters", "handoutMasters"): + wanted += list(self.unpacked_dir.glob(f"ppt/{group}/*.xml")) + wanted += list(self.unpacked_dir.glob(f"ppt/{group}/_rels/*.rels")) + return { + p.relative_to(self.unpacked_dir).as_posix(): p.read_bytes() + for p in wanted + if p.is_file() + } + + def validate_master_theme_uniqueness(self): + from helpers.pptx_theme import _NOTES_MASTERS, live_shared_master_themes + + shared = live_shared_master_themes(self._package_map()) + if shared: + print(f"FAILED - Found {len(shared)} master(s) sharing a theme part:") + for message in shared: + print(f" {message}") + if any(m.startswith(_NOTES_MASTERS) for m in shared): + print(" Fix: in ppt/presentation.xml, move back to " + "directly after . PowerPoint reads that happily.") + else: + print(" Fix: give each master its own theme part.") + return False + + if self.verbose: + print("PASSED - No master shares a theme part in a way PowerPoint refuses") + return True + + def validate_charts(self): + from helpers.pptx_chart import find_chart_problems + + problems = find_chart_problems(self._package_map()) + if problems: + print(f"FAILED - Found {len(problems)} chart problem(s) PowerPoint rejects:") + for message in problems: + print(f" {message}") + return False + + if self.verbose: + print("PASSED - Charts satisfy the constraints PowerPoint enforces") + return True + + def _original_slide_defects(self, schema) -> set[str]: + import tempfile + import zipfile + + from helpers.pptx_slide import SLIDE_PART_RE, fatal_slide_errors + + if self.original_file is None: + return set() + + found: set[str] = set() + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + try: + with zipfile.ZipFile(self.original_file, "r") as zf: + safe_extract(zf, temp_path) + except (zipfile.BadZipFile, ValueError, OSError): + return set() + + for part in sorted(temp_path.rglob("*.xml")): + relative = part.relative_to(temp_path).as_posix() + if not SLIDE_PART_RE.fullmatch(relative): + continue + ok, errors = self._validate_single_file_xsd( + part.resolve(), temp_path.resolve(), schema_path=schema + ) + if ok is None or ok or not errors: + continue + found |= set(fatal_slide_errors(set(errors))) + return found + + def validate_slides(self): + from helpers.pptx_slide import ( + SLIDE_PART_RE, + fatal_slide_errors, + is_schema_verdict, + ) + + schema = self.schemas_dir / self.SCHEMA_MAPPINGS["ppt"] + inherited = self._original_slide_defects(schema) + problems: list[str] = [] + broken: list[str] = [] + + for xml_file in self.xml_files: + relative = xml_file.relative_to(self.unpacked_dir).as_posix() + if not SLIDE_PART_RE.fullmatch(relative): + continue + ok, errors = self._validate_single_file_xsd( + xml_file.resolve(), self.unpacked_dir.resolve(), schema_path=schema + ) + if ok is None or not errors: + continue + + unreadable = [f"{relative}: {e}" for e in errors if not is_schema_verdict(e)] + if unreadable: + broken.extend(unreadable) + continue + if ok: + continue + + for message in fatal_slide_errors(set(errors)): + if message in inherited: + continue + problems.append(f"{relative}: {message}") + + if broken: + print(f"FAILED - Could not check {len(broken)} slide part(s):") + for message in sorted(broken): + print(f" {message[:240]}") + + if problems: + print(f"FAILED - Found {len(problems)} slide problem(s) PowerPoint rejects:") + for message in sorted(problems): + print(f" {message[:240]}") + + if broken or problems: + return False + + if self.verbose: + print("PASSED - Slide XML has none of the defects PowerPoint refuses") + return True + + def _get_schema_path(self, xml_file): + if xml_file.parent.name == "charts" and xml_file.name.startswith("chart"): + return None + return super()._get_schema_path(xml_file) + + def _preprocess_for_schema(self, xml_doc, relative_path): + if relative_path.as_posix() != "ppt/presentation.xml": + return xml_doc + + root = xml_doc.getroot() + ns = f"{{{self.PRESENTATIONML_NAMESPACE}}}" + notes = root.find(f"{ns}notesMasterIdLst") + slides = root.find(f"{ns}sldIdLst") + if notes is None or slides is None: + return xml_doc + + children = list(root) + if children.index(notes) < children.index(slides): + return xml_doc + + root.remove(notes) + root.insert(list(root).index(slides), notes) + return xml_doc + + def validate_uuid_ids(self): + import lxml.etree + + errors = [] + uuid_pattern = re.compile( + r"^[\{\(]?[0-9A-Fa-f]{8}-?[0-9A-Fa-f]{4}-?[0-9A-Fa-f]{4}-?[0-9A-Fa-f]{4}-?[0-9A-Fa-f]{12}[\}\)]?$" + ) + + for xml_file in self.xml_files: + try: + root = lxml.etree.parse(str(xml_file)).getroot() + + for elem in root.iter(): + for attr, value in elem.attrib.items(): + attr_name = attr.split("}")[-1].lower() + if attr_name == "id" or attr_name.endswith("id"): + if self._looks_like_uuid(value): + if not uuid_pattern.match(value): + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: " + f"Line {elem.sourceline}: ID '{value}' appears to be a UUID but contains invalid hex characters" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {xml_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} UUID ID validation errors:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All UUID-like IDs contain valid hex values") + return True + + def _looks_like_uuid(self, value): + clean_value = value.strip("{}()").replace("-", "") + return len(clean_value) == 32 and all(c.isalnum() for c in clean_value) + + def validate_slide_layout_ids(self): + import lxml.etree + + errors = [] + + slide_masters = list(self.unpacked_dir.glob("ppt/slideMasters/*.xml")) + + if not slide_masters: + if self.verbose: + print("PASSED - No slide masters found") + return True + + for slide_master in slide_masters: + try: + root = lxml.etree.parse(str(slide_master)).getroot() + + rels_file = slide_master.parent / "_rels" / f"{slide_master.name}.rels" + + if not rels_file.exists(): + errors.append( + f" {slide_master.relative_to(self.unpacked_dir)}: " + f"Missing relationships file: {rels_file.relative_to(self.unpacked_dir)}" + ) + continue + + rels_root = lxml.etree.parse(str(rels_file)).getroot() + + valid_layout_rids = set() + for rel in rels_root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ): + rel_type = rel.get("Type", "") + if "slideLayout" in rel_type: + valid_layout_rids.add(rel.get("Id")) + + for sld_layout_id in root.findall( + f".//{{{self.PRESENTATIONML_NAMESPACE}}}sldLayoutId" + ): + r_id = sld_layout_id.get( + f"{{{self.OFFICE_RELATIONSHIPS_NAMESPACE}}}id" + ) + layout_id = sld_layout_id.get("id") + + if r_id and r_id not in valid_layout_rids: + errors.append( + f" {slide_master.relative_to(self.unpacked_dir)}: " + f"Line {sld_layout_id.sourceline}: sldLayoutId with id='{layout_id}' " + f"references r:id='{r_id}' which is not found in slide layout relationships" + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {slide_master.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print(f"FAILED - Found {len(errors)} slide layout ID validation errors:") + for error in errors: + print(error) + print( + "Remove invalid references or add missing slide layouts to the relationships file." + ) + return False + else: + if self.verbose: + print("PASSED - All slide layout IDs reference valid slide layouts") + return True + + def validate_no_duplicate_slide_layouts(self): + import lxml.etree + + errors = [] + slide_rels_files = list(self.unpacked_dir.glob("ppt/slides/_rels/*.xml.rels")) + + for rels_file in slide_rels_files: + try: + root = lxml.etree.parse(str(rels_file)).getroot() + + layout_rels = [ + rel + for rel in root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ) + if "slideLayout" in rel.get("Type", "") + ] + + if len(layout_rels) > 1: + errors.append( + f" {rels_file.relative_to(self.unpacked_dir)}: has {len(layout_rels)} slideLayout references" + ) + + except Exception as e: + errors.append( + f" {rels_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + if errors: + print("FAILED - Found slides with duplicate slideLayout references:") + for error in errors: + print(error) + return False + else: + if self.verbose: + print("PASSED - All slides have exactly one slideLayout reference") + return True + + def validate_notes_slide_references(self): + import lxml.etree + + errors = [] + notes_slide_references = {} + + slide_rels_files = list(self.unpacked_dir.glob("ppt/slides/_rels/*.xml.rels")) + + if not slide_rels_files: + if self.verbose: + print("PASSED - No slide relationship files found") + return True + + for rels_file in slide_rels_files: + try: + root = lxml.etree.parse(str(rels_file)).getroot() + + for rel in root.findall( + f".//{{{self.PACKAGE_RELATIONSHIPS_NAMESPACE}}}Relationship" + ): + rel_type = rel.get("Type", "") + if "notesSlide" in rel_type: + part = opc_target( + rel.get("Target", ""), + rels_source_part(rels_file, self.unpacked_dir), + rel.get("TargetMode", ""), + ) + if part: + slide_name = rels_file.stem.replace( + ".xml", "" + ) + + notes_slide_references.setdefault(part, []).append( + (slide_name, rels_file) + ) + + except (lxml.etree.XMLSyntaxError, Exception) as e: + errors.append( + f" {rels_file.relative_to(self.unpacked_dir)}: Error: {e}" + ) + + for target, references in notes_slide_references.items(): + if len(references) > 1: + slide_names = [ref[0] for ref in references] + errors.append( + f" Notes slide '{target}' is referenced by multiple slides: {', '.join(slide_names)}" + ) + for slide_name, rels_file in references: + errors.append(f" - {rels_file.relative_to(self.unpacked_dir)}") + + if errors: + print( + f"FAILED - Found {len([e for e in errors if not e.startswith(' ')])} notes slide reference validation errors:" + ) + for error in errors: + print(error) + print("Each slide may optionally have its own slide file.") + return False + else: + if self.verbose: + print("PASSED - All notes slide references are unique") + return True + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-xlsx/scripts/office/validators/redlining.py b/.github/skills/anthropic-xlsx/scripts/office/validators/redlining.py new file mode 100644 index 00000000..4185c51f --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/office/validators/redlining.py @@ -0,0 +1,299 @@ +""" +Validator for tracked changes in Word documents. + +Detects untracked edits in word/document.xml: text that differs from the +original without a / wrapper recording it. The tracked changes +that are new relative to the original are undone, and the result is compared +against the original; whatever text still differs was edited without being +tracked. + +Only the document body is compared. Headers, footers, footnotes and endnotes +are separate parts and are not checked. +""" + +import subprocess +import tempfile +import zipfile +from pathlib import Path + +import defusedxml.ElementTree as ET +from defusedxml.common import DefusedXmlException + +from helpers import rendered_text, safe_extract + + +class RedliningValidator: + + def __init__(self, unpacked_dir, original_docx, verbose=False): + self.unpacked_dir = Path(unpacked_dir) + self.original_docx = Path(original_docx) + self.verbose = verbose + self.namespaces = { + "w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + } + + def repair(self) -> int: + return 0 + + def validate(self): + modified_file = self.unpacked_dir / "word" / "document.xml" + if not modified_file.exists(): + print(f"FAILED - Modified document.xml not found at {modified_file}") + return False + + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + + try: + with zipfile.ZipFile(self.original_docx, "r") as zip_ref: + safe_extract(zip_ref, temp_path) + except Exception as e: + print(f"FAILED - Error unpacking original docx: {e}") + return False + + original_file = temp_path / "word" / "document.xml" + if not original_file.exists(): + print( + f"FAILED - Original document.xml not found in {self.original_docx}" + ) + return False + + try: + modified_tree = ET.parse(modified_file) + modified_root = modified_tree.getroot() + original_tree = ET.parse(original_file) + original_root = original_tree.getroot() + except (ET.ParseError, DefusedXmlException) as e: + print(f"FAILED - Error parsing XML files: {e}") + return False + + new_changes = self._new_tracked_changes(original_root, modified_root) + self._remove_tracked_changes(modified_root, new_changes) + + modified_text = self._extract_text_content(modified_root) + original_text = self._extract_text_content(original_root) + + if modified_text != original_text: + error_message = self._generate_detailed_diff( + original_text, modified_text + ) + print(error_message) + return False + + if self.verbose: + print( + f"PASSED - All {len(new_changes)} change(s) against the original " + "are properly tracked" + ) + return True + + def _tracked_change_elements(self, root): + ins_tag = f"{{{self.namespaces['w']}}}ins" + del_tag = f"{{{self.namespaces['w']}}}del" + return [elem for elem in root.iter() if elem.tag in (ins_tag, del_tag)] + + def _rendered_text(self, elem): + preserve = elem.get("{http://www.w3.org/XML/1998/namespace}space") == "preserve" + return rendered_text(elem.text or "", preserve) + + def _text_elements(self, elem): + w = self.namespaces["w"] + return [ + node + for node in elem.iter() + if node.tag in (f"{{{w}}}t", f"{{{w}}}delText") + ] + + def _tracked_change_key(self, elem): + w = self.namespaces["w"] + text = "".join(self._rendered_text(node) for node in self._text_elements(elem)) + return (elem.tag, elem.get(f"{{{w}}}author"), elem.get(f"{{{w}}}date"), text) + + def _new_tracked_changes(self, original_root, modified_root): + original = self._tracked_change_elements(original_root) + modified = self._tracked_change_elements(modified_root) + + pool = {} + for elem in original: + pool.setdefault(self._tracked_change_key(elem), []).append(elem) + + matched, leftover = set(), [] + for elem in modified: + bucket = pool.get(self._tracked_change_key(elem)) + if bucket: + matched.add(bucket.pop()) + else: + leftover.append(elem) + + def group(elem): + return self._tracked_change_key(elem)[:3] + + def text_of(elems): + return "".join(self._tracked_change_key(e)[3] for e in elems) + + unmatched_original = {} + for elem in original: + if elem not in matched: + unmatched_original.setdefault(group(elem), []).append(elem) + + by_group = {} + for elem in leftover: + by_group.setdefault(group(elem), []).append(elem) + + new = set() + for key, elems in by_group.items(): + rebuilt = text_of(elems) + if rebuilt and rebuilt == text_of(unmatched_original.get(key, [])): + continue + new.update(elems) + return new + + def _generate_detailed_diff(self, original_text, modified_text): + error_parts = [ + "FAILED - Document text doesn't match after removing the tracked changes", + "", + "Likely causes:", + " 1. Modified text inside another author's or tags", + " 2. Made edits without proper tracked changes", + " 3. Didn't nest inside when deleting another's insertion", + " 4. Rewrote another author's / and changed its text on", + " the way. A tracked change from the original is recognised by its", + " author, date and text; anything that doesn't reproduce one exactly", + " reads as new, and the text it carried is reported missing.", + "", + "For pre-redlined documents, use correct patterns:", + " - To reject another's INSERTION: Nest inside their ", + " - To reject PART of one: nest around only the runs you reject.", + " Their may be split around it, so long as the pieces keep", + " their author and date and still spell out the same text.", + " - To restore another's DELETION: Add new AFTER their ", + "", + ] + + git_diff = self._get_git_word_diff(original_text, modified_text) + if git_diff: + error_parts.extend(["Differences:", "============", git_diff]) + else: + error_parts.append("Unable to generate word diff (git not available)") + + return "\n".join(error_parts) + + def _get_git_word_diff(self, original_text, modified_text): + try: + with tempfile.TemporaryDirectory() as temp_dir: + temp_path = Path(temp_dir) + + original_file = temp_path / "original.txt" + modified_file = temp_path / "modified.txt" + + original_file.write_text(original_text, encoding="utf-8") + modified_file.write_text(modified_text, encoding="utf-8") + + result = subprocess.run( + [ + "git", + "diff", + "--word-diff=plain", + "--word-diff-regex=.", + "-U0", + "--no-index", + str(original_file), + str(modified_file), + ], + capture_output=True, + text=True, + ) + + if result.stdout.strip(): + lines = result.stdout.split("\n") + content_lines = [] + in_content = False + for line in lines: + if line.startswith("@@"): + in_content = True + continue + if in_content and line.strip(): + content_lines.append(line) + + if content_lines: + return "\n".join(content_lines) + + result = subprocess.run( + [ + "git", + "diff", + "--word-diff=plain", + "-U0", + "--no-index", + str(original_file), + str(modified_file), + ], + capture_output=True, + text=True, + ) + + if result.stdout.strip(): + lines = result.stdout.split("\n") + content_lines = [] + in_content = False + for line in lines: + if line.startswith("@@"): + in_content = True + continue + if in_content and line.strip(): + content_lines.append(line) + return "\n".join(content_lines) + + except (subprocess.CalledProcessError, FileNotFoundError, Exception): + pass + + return None + + def _remove_tracked_changes(self, root, targets): + ins_tag = f"{{{self.namespaces['w']}}}ins" + del_tag = f"{{{self.namespaces['w']}}}del" + + for parent in root.iter(): + to_remove = [] + for child in parent: + if child.tag == ins_tag and child in targets: + to_remove.append(child) + for elem in to_remove: + parent.remove(elem) + + deltext_tag = f"{{{self.namespaces['w']}}}delText" + t_tag = f"{{{self.namespaces['w']}}}t" + + for parent in root.iter(): + to_process = [] + for child in parent: + if child.tag == del_tag and child in targets: + to_process.append((child, list(parent).index(child))) + + for del_elem, del_index in reversed(to_process): + for elem in del_elem.iter(): + if elem.tag == deltext_tag: + elem.tag = t_tag + + for child in reversed(list(del_elem)): + parent.insert(del_index, child) + parent.remove(del_elem) + + def _extract_text_content(self, root): + p_tag = f"{{{self.namespaces['w']}}}p" + t_tag = f"{{{self.namespaces['w']}}}t" + + paragraphs = [] + for p_elem in root.findall(f".//{p_tag}"): + text_parts = [] + for t_elem in p_elem.findall(f".//{t_tag}"): + text_parts.append(self._rendered_text(t_elem)) + paragraph_text = "".join(text_parts) + if paragraph_text: + paragraphs.append(paragraph_text) + + return "\n".join(paragraphs) + + +if __name__ == "__main__": + raise RuntimeError("This module should not be run directly.") diff --git a/.github/skills/anthropic-xlsx/scripts/recalc.py b/.github/skills/anthropic-xlsx/scripts/recalc.py new file mode 100644 index 00000000..6232be2d --- /dev/null +++ b/.github/skills/anthropic-xlsx/scripts/recalc.py @@ -0,0 +1,308 @@ +""" +Excel Formula Recalculation Script +Recalculates all formulas in an Excel file using LibreOffice +""" + +import contextlib +import json +import os +import platform +import re +import shutil +import subprocess +import sys +import tempfile +import time +import zipfile +from pathlib import Path + +from office.soffice import get_soffice_env, run_soffice + +from openpyxl import load_workbook + +MACRO_FILENAME = "Module1.xba" +SOFFICE_MISSING = "soffice not found on PATH; LibreOffice is required to recalculate" + +MAX_LOCATIONS = 100 + +EXTERNAL_REF_RE = re.compile(r"""(? + + + Sub RecalculateAndSave() + ThisComponent.calculateAll() + ThisComponent.store() + ThisComponent.close(True) + End Sub +""" + + +def has_gtimeout(): + try: + subprocess.run( + ["gtimeout", "--version"], capture_output=True, timeout=1, check=False + ) + return True + except (FileNotFoundError, subprocess.TimeoutExpired): + return False + + +def _stamp(path): + st = os.stat(path) + return st.st_mtime_ns, st.st_size + + +def setup_libreoffice_macro(profile_dir: Path, timeout=30): + url = profile_dir.as_uri() + try: + run_soffice( + ["--headless", "--terminate_after_init", f"-env:UserInstallation={url}"], + capture_output=True, + timeout=timeout, + ) + except FileNotFoundError: + return None, SOFFICE_MISSING + except subprocess.TimeoutExpired: + return None, "LibreOffice timed out creating its profile; formulas were NOT recalculated" + + macro_dir = profile_dir / "user" / "basic" / "Standard" + if not macro_dir.exists(): + return None, "LibreOffice did not create a usable profile; formulas were NOT recalculated" + + try: + (macro_dir / MACRO_FILENAME).write_text(RECALCULATE_MACRO) + except OSError as e: + return None, f"Could not install the recalculation macro: {e}" + + return url, None + + +def external_links_at_risk(filename): + try: + with zipfile.ZipFile(filename) as archive: + names = archive.namelist() + except (zipfile.BadZipFile, OSError): + return [] + if not any(n.startswith("xl/externalLinks/") for n in names): + return [] + + with contextlib.ExitStack() as stack: + formulas = load_workbook(filename, data_only=False) + stack.callback(formulas.close) + values = load_workbook(filename, data_only=True) + stack.callback(values.close) + + external_names = [ + name + for name, dn in formulas.defined_names.items() + if isinstance(getattr(dn, "value", None), str) and EXTERNAL_REF_RE.search(dn.value) + ] + name_re = ( + re.compile(r"\b(" + "|".join(re.escape(n) for n in external_names) + r")\b") + if external_names + else None + ) + + at_risk = [] + for sheet in formulas.sheetnames: + ws = formulas[sheet] + if not hasattr(ws, "iter_rows"): + continue + cached = values[sheet] + for row in ws.iter_rows(): + for cell in row: + v = cell.value + if not (isinstance(v, str) and v.startswith("=")): + continue + reaches_out = EXTERNAL_REF_RE.search(v) or (name_re and name_re.search(v)) + if reaches_out and cached[cell.coordinate].value is None: + at_risk.append(f"{sheet}!{cell.coordinate}") + return at_risk + + +def recalc(filename, timeout=30, force=False): + if not Path(filename).exists(): + return {"error": f"File {filename} does not exist"} + + abs_path = str(Path(filename).absolute()) + + if not os.access(abs_path, os.W_OK): + return {"error": f"{filename} is not writable; recalculation rewrites the file in place"} + + try: + get_soffice_env() + except Exception as e: + return {"error": f"Could not prepare the LibreOffice environment: {e}"} + + if not force: + try: + at_risk = external_links_at_risk(filename) + except Exception as e: + return {"error": f"Could not inspect {filename} for external links: {e}"} + if at_risk: + shown = at_risk[:MAX_LOCATIONS] + return { + "error": ( + "Refusing to recalculate: this workbook links to another workbook, and " + f"{len(at_risk)} linked cell(s) have lost their cached value (openpyxl strips " + "these on save). Recalculating would resolve them to #NAME? and delete the " + "external links for good. Copy those cells' values from the original file " + "before saving, or pass --force to accept the loss. Charts and conditional " + "formats can hold external references too, so this list may not be exhaustive." + ), + "external_link_cells": shown, + "external_link_cells_truncated": max(0, len(at_risk) - len(shown)), + } + + with tempfile.TemporaryDirectory( + prefix="recalc-lo-profile-", ignore_cleanup_errors=True + ) as profile_dir: + return _recalc_with_profile(filename, abs_path, timeout, Path(profile_dir)) + + +def _recalc_with_profile(filename, abs_path, timeout, profile_dir: Path): + started = time.monotonic() + profile_url, err = setup_libreoffice_macro(profile_dir, timeout=timeout) + if err: + return {"error": err} + + timeout = max(5, int(timeout - (time.monotonic() - started))) + + before = _stamp(abs_path) + + cmd = [ + "soffice", + "--headless", + "--norestore", + f"-env:UserInstallation={profile_url}", + "vnd.sun.star.script:Standard.Module1.RecalculateAndSave?language=Basic&location=application", + abs_path, + ] + + if platform.system() == "Linux" and shutil.which("timeout"): + cmd = ["timeout", str(timeout)] + cmd + elif platform.system() == "Darwin" and has_gtimeout(): + cmd = ["gtimeout", str(timeout)] + cmd + + timed_out = f"LibreOffice timed out after {timeout}s; formulas were NOT recalculated. Re-run with a longer timeout." + + try: + result = subprocess.run( + cmd, capture_output=True, text=True, env=get_soffice_env(), timeout=timeout + 15 + ) + except subprocess.TimeoutExpired: + return {"error": timed_out} + except FileNotFoundError: + return {"error": SOFFICE_MISSING} + + if result.returncode == 124: + return {"error": timed_out} + + if result.returncode != 0: + detail = (result.stderr or "").strip() or f"soffice exited {result.returncode}" + return {"error": f"LibreOffice failed to recalculate: {detail}"} + + if _stamp(abs_path) == before: + return { + "error": ( + "LibreOffice exited cleanly but never rewrote the file, so nothing was " + "recalculated. Check that no other LibreOffice instance is running, then retry." + ) + } + + try: + wb = load_workbook(filename, data_only=True) + + excel_errors = [ + "#VALUE!", + "#DIV/0!", + "#REF!", + "#NAME?", + "#NULL!", + "#NUM!", + "#N/A", + ] + error_details = {err: [] for err in excel_errors} + total_errors = 0 + + for sheet_name in wb.sheetnames: + ws = wb[sheet_name] + if not hasattr(ws, "iter_rows"): + continue + for row in ws.iter_rows(): + for cell in row: + if cell.value is not None and isinstance(cell.value, str): + for err in excel_errors: + if err in cell.value: + location = f"{sheet_name}!{cell.coordinate}" + error_details[err].append(location) + total_errors += 1 + break + + result = { + "status": "success" if total_errors == 0 else "errors_found", + "total_errors": total_errors, + "error_summary": {}, + } + + for err_type, locations in error_details.items(): + if locations: + entry = {"count": len(locations), "locations": locations[:MAX_LOCATIONS]} + if len(locations) > MAX_LOCATIONS: + entry["locations_truncated"] = len(locations) - MAX_LOCATIONS + result["error_summary"][err_type] = entry + + wb.close() + + wb_formulas = load_workbook(filename, data_only=False) + formula_count = 0 + for sheet_name in wb_formulas.sheetnames: + ws = wb_formulas[sheet_name] + if not hasattr(ws, "iter_rows"): + continue + for row in ws.iter_rows(): + for cell in row: + if ( + cell.value + and isinstance(cell.value, str) + and cell.value.startswith("=") + ): + formula_count += 1 + wb_formulas.close() + + result["total_formulas"] = formula_count + + return result + + except Exception as e: + return {"error": str(e)} + + +def main(): + args = [a for a in sys.argv[1:] if a != "--force"] + force = "--force" in sys.argv[1:] + + if not args: + print("Usage: python recalc.py [timeout_seconds] [--force]") + print("\nRecalculates all formulas in an Excel file using LibreOffice") + print("\nReturns JSON with error details:") + print(" - status: 'success' or 'errors_found'") + print(" - total_errors: Total number of Excel errors found") + print(" - total_formulas: Number of formulas in the file") + print(" - error_summary: Breakdown by error type with locations") + print(" - #VALUE!, #DIV/0!, #REF!, #NAME?, #NULL!, #NUM!, #N/A") + print("\nOn any failure the JSON has an 'error' key and no 'status'.") + print("--force recalculates even when it would destroy external links.") + sys.exit(1) + + filename = args[0] + timeout = int(args[1]) if len(args) > 1 else 30 + + result = recalc(filename, timeout, force=force) + print(json.dumps(result, indent=2)) + sys.exit(1 if "error" in result else 0) + + +if __name__ == "__main__": + main() diff --git a/.github/skills/grill-me/SKILL.md b/.github/skills/grill-me/SKILL.md index ff05fa66..207e836e 100644 --- a/.github/skills/grill-me/SKILL.md +++ b/.github/skills/grill-me/SKILL.md @@ -35,3 +35,18 @@ Do not ask questions that can be answered by exploring the codebase, documentati End by summarizing the resolved decisions, explicit assumptions, and any unresolved questions the user chose to accept or defer. + + +## Local guided-question contract + +This repository-owned contract overrides any earlier instruction to ask one question at a time. + +- Ask all currently known questions in numbered bulk question blocks. +- Use `Question`, `Recommendation`, `Why`, and `Default if accepted` for every + numbered question. +- Make `Recommendation` the suggested answer and `Why` its concrete rationale. +- Keep each question, recommendation, and reason brief, clear, and + decision-ready. +- Put unresolved follow-ups in another numbered block. If only one blocking + question remains, present it as a numbered one-item block. + diff --git a/.github/skills/grill-me/agents/openai.yaml b/.github/skills/grill-me/agents/openai.yaml new file mode 100644 index 00000000..5546f10d --- /dev/null +++ b/.github/skills/grill-me/agents/openai.yaml @@ -0,0 +1,5 @@ +interface: + display_name: "Grill Me" + short_description: "Relentless interview to sharpen a plan or design" +policy: + allow_implicit_invocation: false diff --git a/.github/skills/internal-agent-creator/SKILL.md b/.github/skills/internal-agent-creator/SKILL.md index a9acace7..a78c45c7 100644 --- a/.github/skills/internal-agent-creator/SKILL.md +++ b/.github/skills/internal-agent-creator/SKILL.md @@ -43,6 +43,7 @@ Prefer a singular core-skill architecture for routers and broader command center - Build agents that are easy to route to. - Keep one cohesive operating role per agent. +- Expand vague requests into a bounded requirements and operating contract. - Translate imported agent value into repo-local GitHub Copilot form. - Keep long reusable procedures out of agent bodies. - Keep paired agent, existing core skill, and reference files coherent without duplicating the same subtopic across files. @@ -52,6 +53,8 @@ Prefer a singular core-skill architecture for routers and broader command center - Preserve evidence-first guidance patterns for fast-moving vendor or platform domains without cargo-culting obsolete tool wiring. - Use current GitHub Copilot custom-agent frontmatter deliberately instead of stripping supported properties by default. - Make approval boundaries, auditability, and dangerous-operation gates explicit when an agent or nearby workflow needs them. +- Validate new agent names and paths before writing. +- Make context handoff explicit when a coordinator delegates to a worker. ## Read First @@ -62,6 +65,7 @@ Load these inputs before finalizing an internal agent: - `.github/copilot-instructions.md` for the non-negotiable behavior layer - `references/agent-contract.md` when editing frontmatter, `tools:`, core-skill sections, or subagent controls - `references/agent-template.md` when drafting a new agent from scratch +- `references/requirements-and-persona.md` when the request is vague, the role is new or materially changing, or a coordinator must package context for workers - `references/conversion-checklist.md` when normalizing an imported or legacy agent - `references/design-patterns.md` when broadening, splitting, or strengthening an agent - `references/example-transformations.md` when you need before-and-after conversion examples @@ -129,23 +133,29 @@ kept clear without a new reusable owner, stop and surface that boundary. ## Authoring Workflow -1. Define the operating role in one sentence. - Use behavioral scope, not prestige language. -2. Scan neighboring agents and trigger overlap. +1. Run a proportional requirements gate. + Resolve purpose, route, inputs, actions, invocation boundary, risk, output, and validation. Ask only questions that materially change the contract. +2. Define the operating role and stance in one sentence. + Translate persona language into observable behavior. Do not invent credentials or prestige. +3. Validate a new name and target path. + Keep the canonical identifier aligned and confirm the resolved path stays under `.github/agents/`. +4. Scan neighboring agents and trigger overlap. Compare `description:` lines first and resolve collisions before drafting. -3. Confirm the behavior belongs in an agent. +5. Confirm the behavior belongs in an agent. Stop if the main deliverable is a procedure, prompt, scoped instruction, validator, or doc. -4. If the agent cites an existing core skill, define the split explicitly. +6. If the agent cites an existing core skill, define the split explicitly. Keep route, stance, tool contract, and output shape in the agent; keep deep tables, templates, and long checklists in references. -5. Draft the `description:` before the body. +7. Draft the `description:` before the body. If the routing sentence is vague, the rest of the agent will stay vague. -6. Choose the frontmatter and core-skill strategy intentionally. +8. Choose the frontmatter and core-skill strategy intentionally. Keep `tools:` explicit, core skills rare, and support-skill references out of the agent unless explicitly requested. -7. Normalize imported patterns and remove stale baggage. +9. Normalize imported patterns and remove stale baggage. Preserve the decision model while deleting obsolete runtime-specific scaffolding. -8. Add real boundaries and measurable output expectations. +10. Add real boundaries and measurable output expectations. Non-router agents recommend the better owner when the boundary breaks instead of routing automatically. -9. Validate, de-duplicate, and re-check paired assets. +11. Define context handoff for coordinators and workers. + Package objective, bounded evidence, constraints, output shape, and validation without assuming shared chat history. +12. Validate, de-duplicate, and re-check paired assets. Run repository validation and re-check whether the new agent makes another one redundant or leaves the paired bundle out of sync. ## Capability Translation Rules @@ -154,8 +164,9 @@ When learning from richer upstream agents, keep the signal and drop the scaffolding. Translate tool catalogs to short canonical `tools:` lists, expertise catalogs to route or output rules, governance patterns to approval boundaries, and helper-skill lists to zero skill references or one existing -core skill. Use `references/design-patterns.md` for the detailed translation -map. +core skill. Translate persona claims into operating behavior rather than +fictional credentials. Use `references/design-patterns.md` and +`references/requirements-and-persona.md` for the detailed translation maps. ## Governance And Trust Boundaries @@ -197,10 +208,13 @@ Load `references/design-patterns.md` for command-center structure questions and - Prestige-first descriptions that never say when the agent wins routing. - Imported agents copied with stale frontmatter, obsolete tool ids, or UI-only scaffolding. +- Prestige biographies, invented credentials, or personality traits that do not change behavior. +- Mandatory interviews that ask for information already available in repository evidence. - A skill-list section as a dumping ground for unrelated capabilities. - A `## Core Skill` section with more than one skill. - New `## Mandatory Engine Skills`, `## Optional Support Skills`, or `## Preferred/Optional Skills` sections without explicit legacy-compatibility scope. - Routers or coordinators that classify only and do not produce a delegated result or blocking explanation. +- Coordinator prompts that assume workers can see the parent conversation. - Agent bodies that hide constraints in long narrative prose or duplicate existing core-skill detail. ## Validation @@ -208,6 +222,7 @@ Load `references/design-patterns.md` for command-center structure questions and - Run `scripts/audit_agent_contract.py --root .` when comparing against the live agent catalog. - Run `scripts/measure_skill_bundle_tokens.py --skill-dir .github/skills/internal-agent-creator` after editing this bundle. - Confirm name, route, `tools:`, subagent controls, and output expectations with `references/review-checklist.md`. +- Confirm requirements, operating stance, name and path safety, and context handoff with `references/requirements-and-persona.md`. - Confirm `## Core Skill`, when present, has exactly one existing skill; otherwise confirm no skill-list section exists. - Confirm new agents do not introduce legacy skill headings unless the user explicitly requested legacy compatibility. - Confirm referenced core skills and references stay aligned, and run the closest repository validation after changes that affect agent naming or inventory. diff --git a/.github/skills/internal-agent-creator/references/agent-template.md b/.github/skills/internal-agent-creator/references/agent-template.md index 5d2854e5..5d1c2762 100644 --- a/.github/skills/internal-agent-creator/references/agent-template.md +++ b/.github/skills/internal-agent-creator/references/agent-template.md @@ -15,6 +15,9 @@ tools: ['read', 'search'] # Internal Example +## Role + +State the operating stance in one or two behavioral sentences. ## Routing Rules @@ -130,6 +133,9 @@ Focused instructions for the worker's domain. ## Notes +- Use `## Role` to encode a concise operating stance, not a prestige biography + or invented credentials. Keep personality language only when it changes + decisions, evidence handling, or output. - `## Core Skill` is optional. Add it only when exactly one existing skill owns required reusable logic for the agent. - Do not add `## Mandatory Engine Skills`, `## Optional Support Skills`, or `## Preferred/Optional Skills` to new repository-owned agents. - `## Skill Usage Contract` is an exception for user-approved command centers, not a default template section. @@ -140,3 +146,5 @@ Focused instructions for the worker's domain. - Use `user-invocable: false` for agents that should only be accessible as subagents. - If you can remove a section without losing routing clarity, remove it. - `description:` should describe selection conditions, not prestige or generic expertise. +- Run the requirements gate in `requirements-and-persona.md` before using a + template for a new or materially changed role. diff --git a/.github/skills/internal-agent-creator/references/requirements-and-persona.md b/.github/skills/internal-agent-creator/references/requirements-and-persona.md new file mode 100644 index 00000000..087514d1 --- /dev/null +++ b/.github/skills/internal-agent-creator/references/requirements-and-persona.md @@ -0,0 +1,95 @@ +# Requirements and Persona Contract + +Use this reference when creating an agent from a short request, materially +changing an agent's role, or defining coordinator-to-worker context transfer. + +## Requirements Gate + +Resolve only the inputs that materially change the agent contract: + +- purpose and winning route +- expected inputs and repository evidence +- required actions and the smallest safe tool scope +- direct, coordinator, worker, or command-center role +- user invocation and subagent invocation boundaries +- risky operations, approval gates, and audit needs +- role-specific output and validation expectations + +Inspect repository evidence before asking the user for facts that are already +available. Ask one focused question at a time when an unresolved choice would +materially change the result. Do not force an interview when the request and +local contract already make the answer deterministic. + +End the gate with one sentence that states the target role, its main boundary, +and the observable result it must produce. + +## Persona Translation + +Translate a vague persona request into observable behavior, not a fictional +biography. + +- **Identity:** State the operating role and the problem it owns. +- **Expertise:** Name only domains that change routing, evidence selection, or + decisions. +- **Working style:** Define how it inspects, decides, communicates, and stops. +- **Output shape:** Make required results and validation status observable. +- **Constraints:** State what it must not do and the better owner when it loses. +- **Quality bar:** Define the checks that distinguish complete work from a + plausible-looking response. + +Keep personality traits only when they change useful behavior, such as direct +severity labels in a review or cautious evidence handling in a production +workflow. Do not invent credentials, years of experience, authority, or +expertise that the contract cannot substantiate. Avoid prestige language. + +Keep the operating stance concise. Put reusable procedures, large checklists, +and domain handbooks in an existing owner or a bundle-local reference instead +of expanding the agent body. + +## Name and Path Safety + +For a new internal agent: + +1. Require a canonical identifier matching + `^internal-[a-z0-9]+(?:-[a-z0-9]+)*$`. +2. Keep the identifier aligned across the filename stem, frontmatter `name:`, + and command identifier. +3. Resolve the target and verify that it stays under `.github/agents/`. +4. Reject path separators, dot segments, whitespace, shell metacharacters, and + YAML metacharacters. +5. Ask for a safe replacement when a supplied name is suspicious. Do not + silently sanitize it. + +Treat an existing target as an edit. Do not overwrite or replace it until its +current contract and user-owned changes have been inspected. + +## Context Handoff + +Do not assume a subagent can see the parent conversation. A coordinator must +package enough task-local context for the worker to act without reconstructing +hidden intent: + +- objective and bounded scope +- relevant paths, snippets, or evidence +- applicable constraints and approval boundaries +- expected output shape +- validation required before completion + +Pass raw task evidence rather than the coordinator's intended conclusion. +Exclude unrelated conversation history, secrets, and redundant repository +context. Require the worker to report unresolved gaps instead of guessing. + +For repeated routing across several related workers, keep one explicit +coordinator contract and an allowlist in `agents:`. Do not create a companion +routing skill per worker merely for symmetry. + +## Proportional Quality Gate + +Before finalizing, confirm: + +- the route is more specific than the persona label +- the tool contract supports the declared actions and no more +- the working style is actionable without becoming a long procedure +- the output contract exposes missing evidence and validation gaps +- safety rules are task-shaped rather than copied generic boilerplate +- coordinator handoffs include the minimum sufficient context diff --git a/.github/skills/internal-agent-creator/references/review-checklist.md b/.github/skills/internal-agent-creator/references/review-checklist.md index 729c6965..430e49f8 100644 --- a/.github/skills/internal-agent-creator/references/review-checklist.md +++ b/.github/skills/internal-agent-creator/references/review-checklist.md @@ -18,6 +18,14 @@ Use this checklist before finalizing a new or revised internal agent. - Are unrelated responsibilities forcing `and/or` language into the route? - Does any large procedure make the edit out of scope for an agent-only workflow? +## Requirements and Operating Stance + +- Was the requirements gate proportional to the ambiguity and risk? +- Does the role describe observable behavior instead of a prestige biography? +- Are expertise and personality statements limited to behavior that changes the result? +- Are approval, stopping, and validation boundaries explicit where needed? +- For a new agent, was the canonical name validated and the resolved target kept under `.github/agents/`? + ## Core Skill Section Contract - Are the skill identifiers exact and canonical? @@ -54,6 +62,8 @@ Use this checklist before finalizing a new or revised internal agent. - If `agents:` is present, is `agent` included in `tools:`? - Are `handoffs` used only for user-visible sequential transitions, not for autonomous within-turn delegation? - Has `references/subagent-patterns.md` been consulted for orchestration design? +- Does each coordinator-to-worker context handoff include objective, bounded evidence, constraints, expected output, and validation? +- Does the handoff avoid assuming that the worker can see the parent conversation? ## Platform Verification diff --git a/.github/skills/internal-agent-creator/references/subagent-patterns.md b/.github/skills/internal-agent-creator/references/subagent-patterns.md index 2d4d1cfe..d1970b4e 100644 --- a/.github/skills/internal-agent-creator/references/subagent-patterns.md +++ b/.github/skills/internal-agent-creator/references/subagent-patterns.md @@ -73,7 +73,7 @@ Focused instructions for domain A work. ### Router with explicit dispatch targets ```yaml -agents: ['internal-gateway-idea', 'internal-gateway-review', 'internal-gateway-critical-master', 'internal-gateway-simple-task'] +agents: ['internal-gateway-idea', 'internal-gateway-review-generic', 'internal-gateway-critical-master', 'internal-gateway-simple-task'] ``` Only these four agents can be invoked as subagents. The platform enforces this. @@ -117,7 +117,7 @@ Handoffs create guided sequential workflows with user-visible buttons between ag ```yaml handoffs: - label: Start Implementation - agent: internal-gateway-review + agent: internal-gateway-review-generic prompt: Implement the plan outlined above. send: false ``` diff --git a/.github/skills/internal-agent-support-lane-change-engine/SKILL.md b/.github/skills/internal-agent-support-lane-change-engine/SKILL.md index 4713ccc6..ef339910 100644 --- a/.github/skills/internal-agent-support-lane-change-engine/SKILL.md +++ b/.github/skills/internal-agent-support-lane-change-engine/SKILL.md @@ -34,5 +34,4 @@ agents. | `internal-gateway-simple-task` | Planning or governance becomes dominant | `internal-gateway-idea` | | `internal-gateway-simple-task` | Assumption pressure-testing becomes dominant | `internal-gateway-critical-master` | | `internal-gateway-critical-master` | The next step is planning | `internal-gateway-idea` | -| `internal-gateway-idea` | A retained `compact` plan is approved for execution | `internal-gateway-simple-task` | -| `internal-gateway-idea` | A retained `extended` plan is approved for execution | `internal-gateway-execute-plans` | +| `internal-gateway-idea` | An approved retained plan is ready for execution | `internal-gateway-execute-plans` | diff --git a/.github/skills/internal-aws-governance/SKILL.md b/.github/skills/internal-aws-governance/SKILL.md index 92814b60..984f4daa 100644 --- a/.github/skills/internal-aws-governance/SKILL.md +++ b/.github/skills/internal-aws-governance/SKILL.md @@ -5,98 +5,47 @@ description: Use when the user needs AWS governance guidance for IAM operating m # Internal AWS Governance -## Referenced skills +Owns AWS IAM, trust, SCP, federation, permission-boundary, and access-guardrail decisions after the broad structure is known. Separates org-level guardrails from account-level grants and keeps permission decisions auditable. -- `internal-aws-strategic`: route back when direction or tradeoff framing is still unsettled. -- `internal-aws-organization-structure`: route when account, OU, delegated admin, or topology structure is the main decision. -- `internal-aws-operations`: route when rollout validation, monitoring, or evidence is the main need. -- `internal-aws-mcp-research`: load when current AWS IAM or service documentation can change the answer. - -Use this skill when the next need is to define or review AWS identity, access, and guardrail decisions. - -This skill owns governance logic after the broad structure is known. It helps separate org-level guardrails from account-level access design and keeps permission decisions auditable. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-aws`. ## When to use -- The user needs IAM model guidance across accounts. -- The user needs SCP, trust, federation, or permission-boundary guidance. -- The user needs to separate preventive controls from granted permissions. -- The user needs a review of guardrail design, exception handling, or access governance. - -## When not to use - -- The main problem is account or OU layout. -- The main problem is strategic option framing before the governance surface is clear. -- The main problem is monitoring, reporting, backup, or post-rollout validation. -- The task is implementation-only. - -## Main domains covered - -- IAM operating model -- role and group strategy -- trust policy boundaries -- permission boundaries and session constraints -- federation and role assumption patterns -- SCP and tag-policy guardrails -- exception and break-glass handling at governance level -- security guardrails tied to identity and access decisions +- IAM operating-model guidance across accounts. +- SCP, trust, federation, or permission-boundary guidance. +- Separating preventive controls from granted permissions. +- Guardrail design, exception handling, or access-governance review. ## Core rules - Keep org-level guardrails distinct from account-level grants. - Treat SCPs as limits on maximum permission, not as grants. -- Prefer roles and federation over long-lived IAM users unless there is a proven reason not to. +- Prefer roles and federation over long-lived IAM users unless a proven reason exists. - Make scope explicit: root, OU, account set, or single account. - Make exception handling explicit when a control is not universal. -- When AWS returns `AccessDenied` with \"explicit deny in a service control policy\", treat the SCP as the blocking control and inspect its exception scope; target-account IAM role grants cannot override an explicit SCP deny. - -Load `references/guardrail-map.md` when the correct governance surface is ambiguous or when the user needs a deeper split between IAM, trust, SCP, and boundary controls. - -## Use of current facts - -Use `internal-aws-mcp-research` when the answer depends on current AWS IAM semantics, service support for delegated patterns, policy simulation, or current documentation. - -## Output expectations - -For narrow asks, return: - -- recommended governance mechanism -- short reason -- main risk or simulation note - -For broader asks, return: +- On `AccessDenied` with "explicit deny in a service control policy", treat the SCP as the blocking control; target-account IAM grants cannot override an explicit SCP deny. -- governance objective -- scope -- candidate mechanisms -- recommended control stack -- exception or blast-radius note -- what should be validated before rollout +Load `references/guardrail-map.md` when the governance surface is ambiguous or a deeper split between IAM, trust, SCP, and boundary controls is needed. -## Relationship to adjacent skills +## Domains -- `internal-aws-strategic` - Use first when the user still needs option framing or lens selection. -- `internal-aws-organization-structure` - Use when the governance question is actually about where a capability should live. -- `internal-aws-operations` - Use when the next need is preflight, reporting, validation, or operational evidence after the governance design is chosen. +IAM operating model · role and group strategy · trust-policy boundaries · permission boundaries and session constraints · federation and role assumption · SCP and tag-policy guardrails · exception and break-glass handling · security guardrails tied to identity and access. ## Common mistakes | Mistake | Why it matters | Instead | -| --- | --- | --- | -| Using SCPs as if they grant access | Preventive controls get mistaken for execution permissions | Pair SCP guidance with the required IAM grant path and keep their roles distinct | -| Answering a governance question without naming scope | Root, OU, and account-level controls behave very differently | State the exact governance scope before recommending a mechanism | -| Mixing org-wide guardrails and in-account authorization into one vague recommendation | Reviewers cannot see which control prevents versus grants | Separate the org-level mechanism from the account-level authorization design | -| Proposing break-glass access without boundaries or audit expectations | Emergency access becomes a standing privilege with weak accountability | Define who can invoke it, how it is bounded, and what audit evidence must exist | -| Recommending rollout without simulation or staged validation when the blast radius is high | A wide deny or trust failure can interrupt platform operations | Use simulation, targeted rollout, and explicit rollback triggers before widening scope | -| Treating permission boundaries as a replacement for trust design | Delegation is still too broad even if identity policies are constrained | Use permission boundaries to limit delegated builders and trust policies to control who can assume the role | +|---|---|---| +| Using SCPs as if they grant access | Preventive controls mistaken for execution permissions | Pair SCP guidance with the required IAM grant path | +| Answering without naming scope | Root, OU, and account controls behave differently | State the exact governance scope before recommending a mechanism | +| Mixing org-wide guardrails and in-account authorization into one vague recommendation | Reviewers cannot see which control prevents versus grants | Separate the org-level mechanism from the account-level design | +| Proposing break-glass access without boundaries or audit expectations | Emergency access becomes standing privilege with weak accountability | Define who can invoke it, how it is bounded, and what audit evidence must exist | +| Recommending rollout without simulation when blast radius is high | A wide deny or trust failure can interrupt platform operations | Use simulation, targeted rollout, and explicit rollback triggers before widening | +| Treating permission boundaries as a replacement for trust design | Delegation stays too broad even if identity policies are constrained | Use boundaries to limit delegated builders and trust policies to control who assumes the role | ## Validation -- Confirm the governance scope is explicit: root, OU, account set, or single account. -- Confirm the recommended mechanism is clear about whether it prevents, grants, or constrains permissions. -- Confirm trust boundaries and exception paths are explicit when human or workload access crosses account boundaries. -- Confirm staged validation or simulation is named before high-blast-radius rollout. -- Confirm the answer says when operational proof should move to `internal-aws-operations`. +- Governance scope is explicit: root, OU, account set, or single account. +- Recommended mechanism is clear about whether it prevents, grants, or constrains. +- Trust boundaries and exception paths are explicit when access crosses account boundaries. +- Staged validation or simulation is named before high-blast-radius rollout. +- Out-of-scope needs, such as operational proof or structure placement, are identified as outside this lane instead of being answered here. diff --git a/.github/skills/internal-aws-lambda/SKILL.md b/.github/skills/internal-aws-lambda/SKILL.md index 725d1a80..066ffee3 100644 --- a/.github/skills/internal-aws-lambda/SKILL.md +++ b/.github/skills/internal-aws-lambda/SKILL.md @@ -5,70 +5,43 @@ description: Use when designing, implementing, refactoring, or reviewing AWS Lam # Internal AWS Lambda -## Referenced skills +Owns AWS Lambda handler, trigger, packaging, retry, and cold-start behavior after the AWS platform direction is chosen. -- `internal-aws-strategic`: AWS direction or tradeoff questions before implementation. -- `internal-aws-governance`: IAM, trust, queue policy, and guardrail design around Lambda. -- `internal-aws-operations`: rollout validation, monitoring, evidence, recovery, and operational proof. -- `internal-python`: shared Python baseline for Python Lambda code. -- `internal-nodejs`: shared JavaScript, Node.js, and TypeScript baseline for Lambda code. -- `internal-python-project`: structured Python modules used by Lambda handlers. -- `internal-nodejs-project`: structured Node.js or TypeScript modules used by Lambda handlers. -- `internal-terraform`: Terraform infrastructure for Lambda resources. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-aws`. ## When to use -- Implementing or reviewing AWS Lambda handlers in Python, Node.js, or TypeScript. -- Designing API Gateway, Lambda Function URL, or other HTTP-triggered request and response handling. -- Designing SQS-triggered batch processing, retry behavior, DLQ handling, or partial batch failure flows. -- Making packaging, dependency, cold-start, VPC, or runtime-configuration choices that are specific to AWS Lambda behavior. - -## When not to use - -- The main problem is still strategic AWS decision support rather than implementation. -- The main problem is IAM, SCP, trust, or organization-structure design. -- The next need is operational evidence, rollout validation, monitoring posture, or DR validation rather than handler design. -- The task is generic Python or Node.js module design with no AWS Lambda behavior in scope. +- Implementing or reviewing Lambda handlers in Python, Node.js, or TypeScript. +- Designing API Gateway, Function URL, or other HTTP-triggered request/response handling. +- Designing SQS-triggered batch processing, retry, DLQ, or partial-batch-failure flows. +- Packaging, dependency, cold-start, VPC, or runtime-configuration choices specific to Lambda. ## Core guidance -- Keep the Lambda handler as a transport adapter; move business logic to testable helpers or services. -- Make the event source explicit and code to one contract at a time: HTTP, queue, schedule, or another async trigger. -- Parse and validate inputs at the boundary, then normalize the data passed to business logic. -- Initialize AWS clients outside the handler when reuse is safe, but keep imports and dependencies small. -- Prefer modular AWS SDK clients and narrow dependencies over broad convenience packages. -- Size timeout, memory, concurrency, batch size, and queue visibility timeout as one operating profile instead of independent toggles. -- Treat duplicate delivery, retries, and idempotency as normal behavior for asynchronous triggers. -- Use environment variables for configuration only; fetch secrets from managed secret stores. -- Log stable identifiers such as request IDs and message IDs, but do not log raw sensitive payloads by default. +- Keep the handler as a transport adapter; move business logic to testable helpers. +- Code to one event-source contract at a time: HTTP, queue, schedule, or async. +- Parse and validate inputs at the boundary; normalize data passed to business logic. +- Initialize AWS clients outside the handler when reuse is safe; keep imports small. +- Prefer modular SDK clients and narrow dependencies over broad convenience packages. +- Size timeout, memory, concurrency, batch size, and queue visibility timeout as one operating profile. +- Treat duplicate delivery, retries, and idempotency as normal for async triggers. +- Use environment variables for configuration; fetch secrets from managed secret stores. +- Log stable identifiers (request IDs, message IDs); do not log raw sensitive payloads by default. ## Event-source guidance -- **HTTP**: Normalize body, path, and query parsing once; return transport-compatible JSON responses with explicit headers; keep CORS intentional. -- **SQS**: Process records independently, handle poison messages explicitly, and return only failed item identifiers when partial batch retry is enabled. -- **Scheduled events**: Make time-window assumptions explicit and guard against duplicate or overlapping execution. -- **File-driven workflows**: Avoid recursive triggers by separating input and output prefixes or buckets. - -Load `references/examples.md` when you need minimal AWS-specific handler patterns or event-source checklists. +- **HTTP**: normalize body, path, and query once; return transport-compatible JSON with explicit headers; keep CORS intentional. +- **SQS**: process records independently; handle poison messages explicitly; return only failed item identifiers when partial batch retry is enabled. +- **Scheduled**: make time-window assumptions explicit; guard against duplicate or overlapping execution. +- **File-driven**: avoid recursive triggers by separating input and output prefixes or buckets. +Load `references/examples.md` for minimal handler patterns and event-source checklists. Load `references/sharp-edges.md` when diagnosing cold starts, VPC latency, retry storms, response-shape mismatches, or file-ingest recursion. - -## Relationship to adjacent skills - -- Use `internal-aws-strategic` when the AWS direction or tradeoff is still unsettled. -- Use `internal-aws-governance` when the next question is IAM, trust, queue policy, or another guardrail design concern. -- Use `internal-aws-operations` when the next question is rollout validation, monitoring, evidence, or recovery proof rather than implementation. -- Use `internal-python-project` when the Lambda code lives in structured Python application modules. -- Use `internal-nodejs-project` when the Lambda code lives in structured Node.js or TypeScript modules. -- Use `internal-terraform` when the primary change is Terraform infrastructure rather than runtime code. - -## Common mistakes - Load `references/common-mistakes.md` for the full mistake table. ## Validation -- Run unit tests outside the Lambda runtime and mock AWS boundaries. -- For HTTP handlers, test malformed body, path, query, and error-response cases. -- For queue consumers, test duplicate delivery, poison messages, timeout pressure, and partial batch failure behavior. -- Validate code assumptions together with the deployed timeout, memory, event-source, and queue configuration. +- Unit tests run outside the Lambda runtime; AWS boundaries are mocked. +- HTTP handlers: malformed body, path, query, and error-response cases tested. +- Queue consumers: duplicate delivery, poison messages, timeout pressure, and partial batch failure tested. +- Code assumptions validated together with deployed timeout, memory, event-source, and queue configuration. diff --git a/.github/skills/internal-aws-mcp-research/SKILL.md b/.github/skills/internal-aws-mcp-research/SKILL.md index 0c1699cf..02b9f7da 100644 --- a/.github/skills/internal-aws-mcp-research/SKILL.md +++ b/.github/skills/internal-aws-mcp-research/SKILL.md @@ -5,112 +5,63 @@ description: Use when the task needs current AWS documentation or safe IAM inspe # Internal AWS MCP Research -## Referenced skills +Standardizes an AWS research workflow that prefers AWS MCP servers when available and falls back to official AWS documentation. Designed for principal-level platform governance questions, not only application coding. -- `internal-aws-strategic`: route recommendations that affect platform direction or tradeoffs. -- `internal-aws-organization-structure`: route recommendations that affect account, OU, delegated admin, or StackSets topology. -- `internal-aws-governance`: route recommendations that affect SCPs, IAM, trust, or federation. -- `internal-aws-operations`: route recommendations that affect validation, monitoring, backup, or rollout evidence. - -Use this skill when AWS decisions depend on up-to-date documentation or on safe inspection of real IAM state. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-aws`. ## When to use -- The task needs current AWS documentation, regional availability facts, or official AWS guidance that may have changed. -- The task needs safe IAM inspection or policy simulation for Organizations, SCPs, IAM policies, roles, or delegated administrators. -- AWS Knowledge MCP or AWS IAM MCP should be preferred when available, with an AWS-doc fallback when they are not. - -## Purpose - -This skill standardizes an AWS research workflow that prefers AWS MCP servers when available and falls back to official AWS documentation when they are not. - -It is designed for principal-level platform governance questions, not only for application coding. +- Current AWS documentation, regional availability, or official guidance that may have changed. +- Safe IAM inspection or policy simulation for Organizations, SCPs, IAM policies, roles, or delegated administrators. +- AWS Knowledge MCP or AWS IAM MCP should be preferred when available, with an AWS-doc fallback. ## Source priority -1. AWS Knowledge MCP for current AWS documentation, latest guidance, and regional availability -2. AWS IAM MCP in read-only mode for account-specific IAM inspection and policy simulation -3. Official AWS documentation when MCP is unavailable or insufficient +1. AWS Knowledge MCP — current docs, latest guidance, regional availability. +2. AWS IAM MCP (read-only) — account-specific IAM inspection and policy simulation. +3. Official AWS documentation when MCP is unavailable or insufficient. -Do not assume both MCP servers are configured in the current client or session. +Do not assume both MCP servers are configured in the current client. -## Server expectations - -Common server identities: +## Server identities - AWS Knowledge MCP: `aws-knowledge-mcp-server` - AWS IAM MCP: `awslabs.iam-mcp-server` or `iam-mcp-server` -The exact configured name can vary by client. +Exact configured name can vary by client. -## Research workflow +## Workflow 1. Classify the question. - - Documentation, best practices, service behavior, regional support: start with AWS Knowledge MCP - - Real IAM state, principals, attached policies, or permission testing: use AWS IAM MCP - - Mixed questions: use Knowledge MCP first, then IAM MCP for confirmation + - Docs, best practices, service behavior, regional support → Knowledge MCP. + - Real IAM state, principals, attached policies, permission testing → IAM MCP. + - Mixed → Knowledge MCP first, IAM MCP for confirmation. 2. Detect available AWS MCP servers in the current environment. -3. Use the safest tool path first. - - Knowledge MCP for documentation lookup - - IAM MCP in read-only mode for inspection and `simulate_principal_policy` -4. If AWS MCP is unavailable, use official AWS documentation from `references/official-source-map.md`. -5. Summarize the answer with source type clearly labeled: - - AWS docs or Knowledge MCP guidance - - live IAM observation - - inferred recommendation - -## AWS Knowledge MCP usage - -Use AWS Knowledge MCP for: +3. Use the safest tool path first (Knowledge MCP for docs; IAM MCP read-only for inspection and `simulate_principal_policy`). +4. If AWS MCP is unavailable, use `references/official-source-map.md`. +5. Summarize with source type labeled: AWS docs / Knowledge MCP guidance / live IAM observation / inferred recommendation. -- service documentation and API behavior -- best practices and architectural guidance -- latest public AWS guidance -- regional availability checks -- CloudFormation and CDK reference lookups +## Knowledge MCP tool patterns -Prefer these tool patterns when available: +- `search_documentation` — find relevant pages +- `read_documentation` — pull exact page into markdown +- `recommend` — expand from one page to adjacent guidance +- `list_regions`, `get_regional_availability` — region-sensitive design -- `search_documentation` to find relevant pages -- `read_documentation` to pull the exact page into markdown -- `recommend` to expand from one AWS page to adjacent guidance -- `list_regions` and `get_regional_availability` for region-sensitive design +## IAM MCP tool patterns (read-only default) -## AWS IAM MCP usage - -Default to read-only and simulation-oriented work. - -Use AWS IAM MCP for: - -- listing users, roles, groups, and policies -- retrieving attached or inline policy details -- understanding trust relationships -- testing policy effects with `simulate_principal_policy` - -Prefer these operations when available: - -- `list_users` -- `get_user` -- `list_roles` -- `list_groups` -- `get_group` -- `list_policies` -- `get_user_policy` -- `get_role_policy` -- `list_user_policies` -- `list_role_policies` -- `simulate_principal_policy` +- `list_users`, `get_user`, `list_roles`, `list_groups`, `get_group` +- `list_policies`, `get_user_policy`, `get_role_policy`, `list_user_policies`, `list_role_policies` +- `simulate_principal_policy` — test policy effects before proposing rollout ## Safety rules - Treat IAM MCP as read-only by default. -- Do not create, delete, attach, detach, or rotate IAM resources unless the user explicitly asks for a change and the blast radius is understood. -- Prefer `simulate_principal_policy` before proposing policy rollout steps. -- Distinguish clearly between documentation-backed statements and observations from a real AWS account. -- When the answer affects strategic framing, route the recommendation back through `internal-aws-strategic`. -- When the answer affects account layout, delegated admin placement, or StackSets topology, route it through `internal-aws-organization-structure`. -- When the answer affects SCPs, IAM, trust, or federation, route it through `internal-aws-governance`. -- When the answer affects validation, backup, monitoring, or rollout evidence, route it through `internal-aws-operations`. +- Do not create, delete, attach, detach, or rotate IAM resources unless the user explicitly asks and blast radius is understood. +- Prefer `simulate_principal_policy` before proposing policy rollout. +- Distinguish documentation-backed statements from observations of a real AWS account. + +Load `references/mcp-capabilities.md` for capability splits across the two MCP servers. ## Output expectations @@ -121,11 +72,8 @@ Prefer these operations when available: - What remains an architectural recommendation or inference - Safe next steps -## References +## Validation -- `references/official-source-map.md` -- `references/mcp-capabilities.md` -- `internal-aws-strategic` -- `internal-aws-organization-structure` -- `internal-aws-governance` -- `internal-aws-operations` +- Source type (docs / live IAM / inference) is labeled for every claim. +- IAM MCP usage stayed read-only unless an explicit change was requested. +- Implications beyond the research lane are reported as labeled findings, not acted on. diff --git a/.github/skills/internal-aws-mcp-research/references/mcp-capabilities.md b/.github/skills/internal-aws-mcp-research/references/mcp-capabilities.md index e65bb74d..d14e9368 100644 --- a/.github/skills/internal-aws-mcp-research/references/mcp-capabilities.md +++ b/.github/skills/internal-aws-mcp-research/references/mcp-capabilities.md @@ -49,4 +49,4 @@ Operational notes: | "Which regions support this?" | AWS Knowledge MCP | | "What does this role or user currently have?" | AWS IAM MCP | | "Would this policy allow action X on resource Y?" | AWS IAM MCP with simulation | -| "How should we govern this across the org?" | `internal-aws-strategic` first, then `internal-aws-organization-structure` or `internal-aws-governance` plus whichever MCP source supplies the facts | +| "How should we govern this across the org?" | Beyond this research lane; supply the facts from whichever MCP source applies and route back to `internal-aws` when the primary owner cannot be selected from the request | diff --git a/.github/skills/internal-aws-operations/SKILL.md b/.github/skills/internal-aws-operations/SKILL.md index d989a162..9fa066a1 100644 --- a/.github/skills/internal-aws-operations/SKILL.md +++ b/.github/skills/internal-aws-operations/SKILL.md @@ -5,97 +5,46 @@ description: Use when the user needs AWS operational guidance for monitoring, lo # Internal AWS Operations -## Referenced skills +Owns the operational side of the AWS platform: monitoring, evidence, preflight, and post-rollout verification. Does not replace strategic framing, structure design, or governance design. -- `internal-aws-strategic`: route back when the platform direction is still unsettled. -- `internal-aws-organization-structure`: route when account, OU, delegated admin, or topology structure is the main decision. -- `internal-aws-governance`: route when IAM, trust, SCP, or guardrail design is the main decision. -- `internal-aws-mcp-research`: load when current AWS service behavior or IAM inspection can change the validation plan. - -Use this skill when the next need is to validate, observe, or operationalize an AWS platform decision. - -This skill owns the operational side of the platform: monitoring, evidence, preflight, and post-rollout verification. It does not replace strategic framing, structure design, or governance design. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-aws`. ## When to use -- The user needs operational readiness guidance after a design choice. -- The user needs monitoring, logging, backup, restore, or DR validation guidance. -- The user needs preflight or post-rollout validation patterns. -- The user needs reporting, export, or audit-evidence guidance. - -## When not to use - -- The main problem is still choosing the high-level direction. -- The main problem is account, OU, or delegated admin structure. -- The main problem is IAM, SCP, or trust-policy design. -- The task is a narrow implementation change with no operational design question. +- Operational readiness guidance after a design choice. +- Monitoring, logging, backup, restore, or DR validation guidance. +- Preflight or post-rollout validation patterns. +- Reporting, export, or audit-evidence guidance. -## Main domains covered +## Domains -- monitoring and observability posture -- CloudTrail, Config, and audit evidence -- backup and restore expectations -- DR validation and recovery evidence -- preflight checks before rollout -- post-rollout validation -- export and reporting for platform operations -- operational proof that a governance or structure change behaved as expected +Monitoring and observability · CloudTrail, Config, and audit evidence · backup and restore · DR validation and recovery evidence · preflight checks · post-rollout validation · export and reporting · operational proof that a governance or structure change behaved. ## Core rules - Keep validation proportional to blast radius. - Treat backup posture and restore evidence as different things. - Prefer preflight and staged validation before wide rollout when access or platform automation could break. -- Keep monitoring, evidence, and reporting tied to the decision that needs confirmation. +- Tie monitoring, evidence, and reporting to the decision that needs confirmation. - Name what is confirmed, what is inferred, and what still needs a real test. -Load `references/validation-and-evidence.md` when the user needs a deeper checklist for preflight, rollout validation, or DR evidence. - -## Use of current facts - -Use `internal-aws-mcp-research` when the answer depends on current AWS service behavior, current documentation, or safe inspection of real IAM state before rollout. - -## Output expectations - -For narrow asks, return: - -- recommended validation or evidence path -- short reason -- main operational risk - -For broader asks, return: - -- operational objective -- preflight checks -- rollout-stage validation path -- post-rollout evidence path -- recovery or DR note when relevant -- open operational risks - -## Relationship to adjacent skills - -- `internal-aws-strategic` - Use first when the core decision is still unsettled. -- `internal-aws-organization-structure` - Use when the operations question is actually about account, OU, or topology placement. -- `internal-aws-governance` - Use when the operations question is actually about IAM, SCP, trust, or guardrail design rather than validation. +Load `references/validation-and-evidence.md` for a deeper preflight, rollout-validation, or DR-evidence checklist. ## Common mistakes | Mistake | Why it matters | Instead | -| --- | --- | --- | -| Treating monitoring as proof that restore or recovery works | Healthy telemetry does not prove recovery viability | Keep backup posture, restore proof, and DR validation as separate evidence lines | -| Skipping preflight for high-blast-radius rollout | Access, logging, or automation regressions are discovered too late | Define preflight checks, rollback trigger, and owner before rollout starts | -| Reporting only control intent without operational evidence | The platform looks compliant on paper but not in practice | Record what was observed in CloudTrail, Config, logs, or recovery tests | -| Mixing validation advice with new governance design instead of keeping the boundary clear | The answer stops being a reliable operations owner | Keep new guardrail design in `internal-aws-governance` and validate the chosen design here | -| Giving a DR answer without making the business criticality assumption visible | Recovery effort may be overbuilt or underbuilt | State the assumed RTO, RPO, or criticality before recommending the evidence path | +|---|---|---| +| Treating monitoring as proof that restore works | Healthy telemetry does not prove recovery viability | Keep backup posture, restore proof, and DR validation as separate evidence lines | +| Skipping preflight for high-blast-radius rollout | Access, logging, or automation regressions discovered too late | Define preflight checks, rollback trigger, and owner before rollout starts | +| Reporting control intent without operational evidence | The platform looks compliant on paper but not in practice | Record what was observed in CloudTrail, Config, logs, or recovery tests | +| Mixing validation advice with new governance design | The answer stops being a reliable operations owner | Keep new guardrail design out of the validation answer and validate the chosen design here | +| Giving a DR answer without making criticality assumption visible | Recovery effort may be overbuilt or underbuilt | State assumed RTO, RPO, or criticality before recommending the evidence path | | Treating one successful rollout wave as proof for all scopes | Wider OUs, regions, or accounts can still fail differently | Widen only after the first safe unit is validated and recorded | ## Validation -- Confirm the answer distinguishes confirmed evidence from inferred evidence. -- Confirm preflight checks, rollback trigger, and rollout unit are explicit for risky changes. -- Confirm backup proof and restore proof are treated as separate validation paths when state exists. -- Confirm the main operational signals are named for the affected surface, not as a generic checklist. -- Confirm DR or continuity notes are included only when business criticality or recovery posture is actually in scope. +- Confirmed evidence is distinguished from inferred evidence. +- Preflight checks, rollback trigger, and rollout unit are explicit for risky changes. +- Backup proof and restore proof are separate validation paths when state exists. +- Main operational signals are named for the affected surface, not as a generic checklist. +- DR or continuity notes appear only when business criticality or recovery posture is in scope. diff --git a/.github/skills/internal-aws-organization-structure/SKILL.md b/.github/skills/internal-aws-organization-structure/SKILL.md index db35daa4..4b87a170 100644 --- a/.github/skills/internal-aws-organization-structure/SKILL.md +++ b/.github/skills/internal-aws-organization-structure/SKILL.md @@ -5,103 +5,52 @@ description: Use when the user needs AWS control-plane or multi-account structur # Internal AWS Organization Structure -## Referenced skills +Owns AWS layout decisions: account, OU, delegated admin, StackSets topology, and platform-level network layout. Translates a platform goal into account, OU, delegated admin, network, and rollout structure. Does not own generic strategy, detailed IAM, or monitoring implementation. -- `internal-aws-strategic`: route back when the platform direction or tradeoff frame is still unsettled. -- `internal-aws-governance`: route guardrail, IAM, SCP, trust, or federation design. -- `internal-aws-operations`: route monitoring, validation, backup, reporting, or post-rollout evidence. -- `internal-aws-mcp-research`: load when current AWS service support or delegated-admin behavior can change the structure answer. - -Use this skill when the next need is to design or review how AWS is structured at organization and platform level. - -This skill owns AWS layout decisions, not generic strategy and not detailed IAM or monitoring implementation. It helps translate a platform goal into account, OU, delegated admin, network, and rollout structure. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-aws`. ## When to use -- The user is shaping or reviewing AWS Organizations layout. -- The user needs account, OU, or payer-management separation guidance. -- The user is deciding delegated administrator placement. -- The user needs StackSets topology or rollout-scope guidance. -- The user needs network placement or multi-region layout at platform level. - -## When not to use +- Shaping or reviewing AWS Organizations layout. +- Account, OU, or payer-management separation guidance. +- Delegated administrator placement decisions. +- StackSets topology or rollout-scope guidance. +- Network placement or multi-region layout at platform level. -- The question is mainly IAM, SCP, federation, or guardrail logic. -- The task is mainly monitoring, backup, reporting, or post-rollout validation. -- The user only needs generic strategic comparison with no concrete structure question. -- The task is already implementation-focused. +## Domains -## Main domains covered - -- AWS Organizations hierarchy -- OU design and safe rollout scope -- management account versus payer responsibilities -- account segmentation and account purpose -- shared services, security, and log archive account layout -- delegated administrator placement -- StackSets topology and blast radius at structure level -- platform-level network topology -- multi-account and multi-region structural decisions +AWS Organizations hierarchy · OU design and safe rollout scope · management vs payer responsibilities · account segmentation and purpose · shared services, security, and log-archive account layout · delegated administrator placement · StackSets topology and blast radius · platform-level network topology · multi-account and multi-region structural decisions. ## Working model - Keep the management account minimal unless AWS explicitly requires otherwise. - Distinguish financial ownership from operational ownership. - Prefer delegated administration when it materially reduces blast radius. -- Separate structure decisions from governance decisions: - - structure decides where capabilities live - - governance decides what controls and permissions apply +- Separate structure (where capabilities live) from governance (what controls apply). - Name the smallest safe rollout unit for structural change: account, OU, or region set. -## Research and current facts - -Use `internal-aws-mcp-research` when the answer depends on current AWS service support, delegated admin capabilities, StackSets behavior, or region-sensitive platform constraints. - Load `references/control-surface-map.md` for the control-surface split and default review checklist when the structure choice is ambiguous. ## Output expectations -Keep outputs proportional to the question. - -For narrow asks, return: - -- recommended structure choice -- short reason -- main blast-radius or rollout note - -For broader asks, return: - -- structural objective -- candidate layouts -- recommended placement model -- smallest safe rollout unit -- main risks -- what should move next to `internal-aws-governance` or `internal-aws-operations` - -## Relationship to adjacent skills - -- `internal-aws-strategic` - Use first when the user still needs a broader decision framing or lens selection. -- `internal-aws-governance` - Use when the structural decision is accepted and the next need is IAM, SCP, trust, federation, or guardrail definition. -- `internal-aws-operations` - Use when the structure is accepted and the next need is validation, monitoring, backup, or operational evidence. +Narrow asks: recommended structure choice · short reason · main blast-radius or rollout note. +Broader asks: structural objective · candidate layouts · recommended placement model · smallest safe rollout unit · main risks. ## Common mistakes | Mistake | Why it matters | Instead | -| --- | --- | --- | -| Treating the management account as the default operating account | It increases blast radius and weakens separation of duties | Keep the management account minimal and prefer delegated administrator accounts when AWS supports them | -| Mixing payer responsibility with day-to-day operational ownership without making the reason explicit | Finance and platform controls drift together and are harder to change safely | State the financial owner and the operational owner separately | +|---|---|---| +| Treating the management account as the default operating account | Increases blast radius and weakens separation of duties | Keep management account minimal and prefer delegated administrator accounts | +| Mixing payer responsibility with day-to-day operational ownership | Finance and platform controls drift together and are harder to change | State financial owner and operational owner separately | | Proposing OU or account layouts without a rollout scope | Structural changes become hard to stage or roll back | Name the smallest safe rollout unit: account, OU, or region set | | Hiding global-resource or cross-region blast radius in StackSets discussions | Failures spread further than the rollout plan suggests | Make regional scope, global resources, and rollback boundaries explicit | -| Using structure answers to sneak in IAM or SCP design without separating the concerns | The lane boundary blurs and review gets weaker | Keep placement decisions in this skill and hand guardrail logic to `internal-aws-governance` | -| Recommending shared services placement without naming service ownership | Central accounts become generic dumping grounds | State which platform capability lives centrally and which workload teams still own their execution accounts | +| Using structure answers to sneak in IAM or SCP design | Lane boundary blurs and review gets weaker | Keep placement here and keep guardrail logic out of the structure answer | +| Recommending shared services placement without naming ownership | Central accounts become dumping grounds | State which platform capability lives centrally and which workload teams own execution accounts | ## Validation -- Confirm the placement model is explicit: management account, delegated administrator, shared-services account, or member account. -- Confirm the smallest safe rollout unit is named and matches the proposed structural change. -- Confirm blast radius is explicit for OU moves, delegated admin changes, StackSets rollout, or regional topology shifts. -- Confirm financial ownership and operational ownership are separated when both appear in the answer. -- Confirm the next handoff is clear when the user now needs guardrails or operational validation. +- Placement model is explicit: management account, delegated administrator, shared-services account, or member account. +- Smallest safe rollout unit is named and matches the proposed structural change. +- Blast radius is explicit for OU moves, delegated admin changes, StackSets rollout, or regional topology shifts. +- Financial ownership and operational ownership are separated when both appear. +- Out-of-scope needs, such as guardrail design or operational validation, are identified as outside this lane instead of being answered here. diff --git a/.github/skills/internal-aws-organization-structure/references/control-surface-map.md b/.github/skills/internal-aws-organization-structure/references/control-surface-map.md index 825766e0..c983d945 100644 --- a/.github/skills/internal-aws-organization-structure/references/control-surface-map.md +++ b/.github/skills/internal-aws-organization-structure/references/control-surface-map.md @@ -21,7 +21,7 @@ Use this reference when turning a structural AWS question into the right control | Need | Use first | Notes | | --- | --- | --- | -| Shape preventive boundaries across many accounts | OU design plus `internal-aws-governance` | Keep the structure and the guardrail choice separate | +| Shape preventive boundaries across many accounts | OU design | Keep the structure choice here and treat the guardrail mechanism as a separate decision | | Design a central operating account for an AWS service | Delegated admin placement | Use management account only when AWS requires it | | Roll out a baseline stack across many accounts | StackSets topology | Keep global-resource blast radius explicit | | Separate finance oversight from platform execution | payer and management responsibility split | Make the ownership model explicit | diff --git a/.github/skills/internal-aws-strategic/agents/openai.yaml b/.github/skills/internal-aws-strategic/agents/openai.yaml deleted file mode 100644 index f2d2040d..00000000 --- a/.github/skills/internal-aws-strategic/agents/openai.yaml +++ /dev/null @@ -1,4 +0,0 @@ -interface: - display_name: "Internal AWS Strategic" - short_description: "AWS decision framing and tradeoff support" - default_prompt: "Use $internal-aws-strategic to frame this AWS decision, keep the answer proportional, and pull in only the lenses that matter." diff --git a/.github/skills/internal-aws-strategic/SKILL.md b/.github/skills/internal-aws/SKILL.md similarity index 52% rename from .github/skills/internal-aws-strategic/SKILL.md rename to .github/skills/internal-aws/SKILL.md index 3fd9df36..4d451707 100644 --- a/.github/skills/internal-aws-strategic/SKILL.md +++ b/.github/skills/internal-aws/SKILL.md @@ -1,48 +1,58 @@ --- -name: internal-aws-strategic -description: Use when the user needs high-level AWS platform decision support or tradeoff framing before implementation, and the next step is not yet structure, governance, operations, or code delivery. +name: internal-aws +description: Use when an AWS task cannot be routed confidently to a specific AWS skill because the request is materially ambiguous, has multiple AWS domains with no clear primary owner, or requires clarification before selecting the correct specialist, or when the user needs high-level AWS platform decision support or tradeoff framing before implementation. Do not use for clearly scoped organization structure, governance or IAM, operations or validation, Lambda, or current AWS documentation research tasks. --- -# Internal AWS Strategic +# Internal AWS -## Referenced skills - -- `antigravity-aws-cost-optimizer`: cost-analysis depth when AWS cost data or recommendations are primary. -- `internal-aws-organization-structure`: route when account, OU, delegated admin, or topology structure becomes the decision. -- `internal-aws-governance`: route when IAM, trust, SCP, or guardrail design becomes the decision. -- `internal-aws-operations`: route when validation, monitoring, backup, or audit evidence becomes the decision. -- `internal-aws-mcp-research`: load when current AWS documentation or service behavior can change the recommendation. -- `internal-terraform`: route implementation work in Terraform or OpenTofu. -- `internal-python-script`: route implementation work in standalone Python automation. -- `internal-bash-script`: route implementation work in standalone Bash automation. - -Use this skill when the main need is to reason about an AWS decision before implementation. - -This is a strategic support skill. It helps frame the decision, compare realistic options, expose tradeoffs, and recommend a direction. It does not implement the change and it does not choose Terraform, Python, or Bash on behalf of the user. +Fallback router for AWS tasks that cannot be assigned confidently to one specialist, and strategic support skill for high-level AWS decision framing. Do not activate only because the task concerns AWS; activate only when material routing uncertainty blocks owner selection or when the user needs decision support before the next step is structure, governance, operations, or delivery. ## When to use -- The user needs AWS decision support before execution. -- Multiple AWS approaches are credible and tradeoffs matter. -- The user wants a recommendation grounded in current AWS guidance. -- The user wants high-level support for platform, control-plane, governance, resilience, cost, or operational decisions. +- Material ambiguity prevents selecting one primary AWS specialist. +- Multiple AWS domains are material and no primary owner can be identified safely. +- The user explicitly invokes `$internal-aws`. +- The task asks which AWS lane should own the work before requesting a domain solution. +- The user needs high-level AWS decision support or tradeoff framing before implementation. ## When not to use - The task is already a clear implementation change. -- The user only needs Terraform, Python, Bash, IAM, or operations implementation detail. +- The user only needs detailed IAM, SCP, monitoring, backup, or Lambda implementation detail. - The task is purely post-rollout validation or evidence gathering. - The request is narrow and operational with no real decision to frame. -## Main domains covered +## Routing threshold + +Activate only when at least one holds: + +- the request is materially ambiguous and clarification is required before an AWS owner can be selected; +- multiple AWS domains are material and no primary owner can be identified safely; +- the task asks which AWS problem-solving lane should own the work; +- the user needs strategic decision framing and the next step is not yet structure, governance, operations, or delivery. + +Do not activate when one specialist clearly owns the next step; route directly to that specialist instead. Explicit `$internal-aws` invocation remains valid. + +## Handoffs -- platform and control-plane decision framing -- AWS best-practice and design guidance -- organizational and account-model implications at decision level -- governance and identity implications at decision level -- operational implications at decision level -- resilience and continuity implications at decision level -- cost-value and FinOps implications at decision level +| To | Owns | +|---|---| +| `internal-aws-organization-structure` | account, OU, delegated admin, StackSets, platform network topology | +| `internal-aws-governance` | IAM, trust, SCP, federation, guardrails | +| `internal-aws-operations` | monitoring, validation, backup, recovery, reporting, evidence | +| `internal-aws-lambda` | Lambda runtime, handler, trigger, packaging, retry | +| `internal-aws-mcp-research` | current AWS documentation and safe IAM inspection | +| `antigravity-aws-cost-optimizer` | AWS-specific cost analysis when cost data is the primary problem | + +## Dispatch contract + +1. State the routing uncertainty. +2. Identify candidate AWS owners. +3. Select the minimum specialist set. +4. Keep strategic comparison here only while needed to choose the owner. +5. Hand the resolved task to the primary specialist instead of retaining ownership. + +Load `references/routing-matrix.md` for the routing decision tree. Load `references/lens-playbook.md` when the fallback trigger fires and the choice of AWS owner or lens needs structured comparison, or when the user wants a deeper decision-framing aid. ## Optional lens activation @@ -72,8 +82,6 @@ Rules: - If another lens would materially improve the recommendation, suggest it briefly instead of forcing it. - Keep the active lenses explicit when more than one is in play. -Load `references/lens-playbook.md` when the user wants a deeper framing aid or when the choice of lenses is not obvious. - ## Optional BC/DR lens BC/DR is optional. @@ -140,34 +148,25 @@ Include: - main risks and blast radius - validation or follow-up path -## Relationship to adjacent skills - -- `antigravity-aws-cost-optimizer` - Use as depth support when the strategic question becomes cost analysis, spend optimization, or FinOps estimation that needs AWS-specific billing and Cost Explorer guidance. -- `internal-aws-organization-structure` - Use when the next need is account model, OU layout, delegated admin placement, network topology, or control-plane structure. -- `internal-aws-governance` - Use when the next need is IAM model, SCP design, trust boundaries, federation, or guardrail definition. -- `internal-aws-operations` - Use when the next need is monitoring, backup, recovery validation, reporting, audit evidence, preflight, or post-rollout checks. -- `internal-terraform`, `internal-python-script`, `internal-bash-script` - Use when the decision is settled and implementation begins. - ## Common mistakes | Mistake | Why it matters | Instead | | --- | --- | --- | -| Forcing a full multi-lens analysis for a small question | The answer gets heavy without improving the decision | Start with the smallest useful lens set and widen only if risk or ambiguity justifies it | -| Treating BC/DR as mandatory for every answer | Continuity concerns drown out the actual decision | Activate BC/DR only when recovery posture materially changes the recommendation | +| Forcing a full multi-lens analysis for a small question | The answer becomes heavier than the decision requires | Start with the smallest useful lens set and widen only if risk or ambiguity justifies it | +| Treating BC/DR as mandatory for every answer | Continuity language crowds out the actual AWS tradeoff | Activate BC/DR only when recovery posture or multi-region continuity changes the recommendation | | Recommending a direction without current-source verification when freshness matters | AWS support boundaries, limits, or service behavior may have changed | Call out the freshness dependency and route to `internal-aws-mcp-research` when it can change the decision | | Confusing decision support with implementation guidance | The user loses the strategic framing they asked for | Keep the answer at decision level and hand off only after the direction is chosen | | Expanding into tool or IaC selection when the user did not ask for it | The response drifts from AWS platform tradeoffs into execution detail | Keep the recommendation centered on the AWS choice, not the delivery tooling | +| Activating the fallback when one specialist clearly owns the next step | The router delays work a direct specialist should own | Route directly to the specialist and keep the fallback for genuine uncertainty | | Giving generic best-practice advice without context, tradeoff, or cost implication | Generic guidance is hard to act on and easy to misapply | Tie the recommendation to assumptions, viable options, and cost-value consequences | ## Validation +- State why the request could not be assigned to one primary AWS specialist, or name the decision being framed. +- Confirm the selected specialist set is the minimum needed to resolve the uncertainty. - Confirm the decision statement is explicit and narrow enough that the next owner is obvious. - Confirm assumptions, active lenses, and the main tradeoff are named instead of implied. - Confirm the recommendation includes reversibility or blast-radius guidance when the choice is hard to unwind. - Confirm cost-value or operational impact is called out when it materially changes the recommendation. - Confirm the answer states when freshness matters and whether `internal-aws-mcp-research` should be used. +- Confirm the resolved task is handed to a primary specialist. diff --git a/.github/skills/internal-aws/agents/openai.yaml b/.github/skills/internal-aws/agents/openai.yaml new file mode 100644 index 00000000..f9dbe341 --- /dev/null +++ b/.github/skills/internal-aws/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "Internal AWS" + short_description: "AWS routing and strategic decision support" + default_prompt: "Use $internal-aws to route an unclear AWS task to the minimum specialist set, or to frame an AWS decision when the next step is not yet structure, governance, operations, or delivery." diff --git a/.github/skills/internal-aws-strategic/references/lens-playbook.md b/.github/skills/internal-aws/references/lens-playbook.md similarity index 100% rename from .github/skills/internal-aws-strategic/references/lens-playbook.md rename to .github/skills/internal-aws/references/lens-playbook.md diff --git a/.github/skills/internal-aws/references/routing-matrix.md b/.github/skills/internal-aws/references/routing-matrix.md new file mode 100644 index 00000000..ee076b5e --- /dev/null +++ b/.github/skills/internal-aws/references/routing-matrix.md @@ -0,0 +1,37 @@ +# Internal AWS Routing Matrix + +## Fallback-positive cases + +- An underspecified multi-account control problem mixes organization structure, + governance, and operations without identifying a primary deliverable. Use + `internal-aws` to state the uncertainty, identify candidate owners, and + select the minimum specialist set. +- An AWS platform question asks which problem-solving lane should own the work, + but the request does not identify whether the primary concern is topology, + governance, operations, Lambda, or current documentation. Use `internal-aws` + to clarify the lane before dispatch. + +## Direct-specialist negative cases + +- OU layout or account placement → `internal-aws-organization-structure`. +- SCP or permission-boundary design → `internal-aws-governance`. +- Backup and restore validation → `internal-aws-operations`. +- Lambda retry behavior → `internal-aws-lambda`. +- Current AWS documentation lookup → `internal-aws-mcp-research`. + +## Multi-domain primary-owner cases + +- OU design with later SCP work → `internal-aws-organization-structure` first; + hand the resulting governance implications to `internal-aws-governance`. +- SCP rollout evidence → `internal-aws-governance` first and + `internal-aws-operations` second for evidence or validation. +- Lambda IAM detail → `internal-aws-lambda` when the requested deliverable is + Lambda behavior, or `internal-aws-governance` when the requested deliverable + is the IAM or trust boundary. Do not use the fallback when that deliverable + is explicit. + +## Review rule + +Prefer a direct specialist whenever a reasonable reviewer can name one primary +owner from the request itself. The fallback is not a prerequisite for ordinary +AWS work and must never activate all AWS skills by default. diff --git a/.github/skills/internal-azure-devops/SKILL.md b/.github/skills/internal-azure-devops/SKILL.md index d24b9416..c3da1908 100644 --- a/.github/skills/internal-azure-devops/SKILL.md +++ b/.github/skills/internal-azure-devops/SKILL.md @@ -1,18 +1,17 @@ --- name: internal-azure-devops -description: Use when authoring, reviewing, or routing Azure DevOps pipelines or project automation before CLI operations need a narrower owner. +description: Use when the user needs Azure DevOps pipeline YAML authoring or review, project automation, pipeline triggers, variables, environments, approvals, or artifact-flow guidance. Do not use for GitHub Actions; Azure CLI or Resource Manager work with no pipeline surface; tenant, governance, or operations design; or materially ambiguous requests with no clear pipeline deliverable. --- # Internal Azure DevOps -## Referenced skills +## Handoffs -Treat the referenced skills below as on-demand supports. Do not preload them -for every pipeline task; load only the owner proved by YAML structure, CLI -operations, or the active Azure DevOps surface. - -- `internal-yaml`: baseline YAML structure when editing Azure DevOps pipeline YAML. -- `awesome-copilot-azure-devops-cli`: Azure DevOps CLI operations, projects, repos, pipelines, builds, pull requests, work items, artifacts, and service endpoints. +| To | When | +|---|---| +| `internal-azure` | material routing uncertainty prevents selecting a primary Azure specialist | +| `internal-yaml` | baseline YAML structure when editing pipeline YAML | +| `awesome-copilot-azure-devops-cli` | direct CLI operations | ## When to use @@ -25,6 +24,7 @@ operations, or the active Azure DevOps surface. - GitHub Actions workflows or composite actions; use the GitHub Actions owners. - Azure CLI or Azure Resource Manager work with no Azure DevOps project or pipeline surface. - Direct Azure DevOps CLI execution; use `awesome-copilot-azure-devops-cli`. +- The request is materially ambiguous and no primary Azure owner can be named → `internal-azure`. ## Baseline diff --git a/.github/skills/internal-azure-governance/SKILL.md b/.github/skills/internal-azure-governance/SKILL.md index d9825627..cf5a71a7 100644 --- a/.github/skills/internal-azure-governance/SKILL.md +++ b/.github/skills/internal-azure-governance/SKILL.md @@ -1,21 +1,16 @@ --- name: internal-azure-governance -description: Use when the user needs Azure governance guidance for RBAC operating models, managed identity boundaries, PIM or PAM posture, Azure Policy and initiatives, naming and tagging guardrails, exception handling, or other controls that define what principals can do after the Azure structure is chosen. +description: Use when the user needs Azure RBAC operating models, managed identity boundaries, PIM or PAM posture, Azure Policy and initiatives, naming and tagging guardrails, or exception-handling design after the Azure structure is chosen. Do not use for tenant or subscription layout; monitoring, backup, or rollout validation; Azure DevOps pipelines; or materially ambiguous requests with no clear governance deliverable. --- # Internal Azure Governance -## Referenced skills - -- `internal-azure-strategic`: route back when direction or tradeoff framing is still unsettled. -- `internal-azure-organization-structure`: route when tenant, management-group, subscription, or landing-zone layout is the main decision. -- `internal-azure-operations`: route when rollout validation, monitoring, backup, or evidence is the main need. -- `awesome-copilot-azure-role-selector`: least-privilege RBAC role selection depth. - Use this skill when the next need is to define or review Azure identity, access, and guardrail decisions. This skill owns governance logic after the broad structure is known. It helps separate tenant or management-group guardrails from subscription or workload grants and keeps permission decisions auditable. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-azure`. + ## When to use - The user needs RBAC model guidance across management groups or subscriptions. @@ -23,13 +18,6 @@ This skill owns governance logic after the broad structure is known. It helps se - The user needs Azure Policy, initiative, naming, or tagging guardrails. - The user needs a review of guardrail design, exceptions, or access governance. -## When not to use - -- The main problem is management-group, landing-zone, or subscription layout. -- The main problem is strategic option framing before the governance surface is clear. -- The main problem is monitoring, reporting, backup, or post-rollout validation. -- The task is implementation-only. - ## Main domains covered - RBAC operating model @@ -72,17 +60,6 @@ For broader asks, return: - exception or blast-radius note - what should be validated before rollout -## Relationship to adjacent skills - -- `internal-azure-strategic` - Use first when the user still needs option framing or lens selection. -- `internal-azure-organization-structure` - Use when the governance question is actually about where a capability should live. -- `internal-azure-operations` - Use when the next need is preflight, reporting, validation, or operational evidence after the governance design is chosen. -- `awesome-copilot-azure-role-selector` - Use as depth support when the governance boundary is already clear and the next need is least-privilege role selection, custom-role fallback, or assignment artifacts such as CLI commands and Bicep snippets. - ## Common mistakes | Mistake | Why it matters | Instead | @@ -100,4 +77,4 @@ For broader asks, return: - Confirm the recommended mechanism is clear about whether it prevents, grants, or constrains privileged access. - Confirm identity boundaries and exception paths are explicit for human and workload access. - Confirm staged rollout validation is named before high-blast-radius Policy, RBAC, or PIM changes. -- Confirm the answer says when operational proof should move to `internal-azure-operations` and when least-privilege role depth should move to `awesome-copilot-azure-role-selector`. +- Confirm out-of-scope needs, such as operational proof or structure placement, are identified as outside this lane instead of being answered here. diff --git a/.github/skills/internal-azure-operations/SKILL.md b/.github/skills/internal-azure-operations/SKILL.md index c5cf78ab..57bf174b 100644 --- a/.github/skills/internal-azure-operations/SKILL.md +++ b/.github/skills/internal-azure-operations/SKILL.md @@ -1,21 +1,16 @@ --- name: internal-azure-operations -description: Use when the user needs Azure operational guidance for monitoring, logging, backup and restore, Site Recovery or DR validation, preflight checks, post-rollout validation, reporting, or audit evidence after a structure or governance decision has already been made. +description: Use when the user needs Azure monitoring and logging posture, backup and restore proof, Site Recovery or DR validation, preflight checks, post-rollout validation, reporting, or audit evidence after a structure or governance decision is made. Do not use for tenant or subscription layout; RBAC, Policy, or identity design; Azure DevOps pipelines; or materially ambiguous requests with no clear operational deliverable. --- # Internal Azure Operations -## Referenced skills - -- `internal-azure-strategic`: route back when direction or tradeoff framing is still unsettled. -- `internal-azure-organization-structure`: route when tenant, management-group, subscription, or landing-zone layout is the main decision. -- `internal-azure-governance`: route when RBAC, managed identity, PIM, Policy, or guardrail design is the main decision. -- `awesome-copilot-azure-resource-health-diagnose`: Azure resource-health diagnosis depth. - Use this skill when the next need is to validate, observe, or operationalize an Azure platform decision. This skill owns the operational side of the platform: monitoring, evidence, preflight, and post-rollout verification. It does not replace strategic framing, structure design, or governance design. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-azure`. + ## When to use - The user needs operational readiness guidance after a design choice. @@ -23,13 +18,6 @@ This skill owns the operational side of the platform: monitoring, evidence, pref - The user needs preflight or post-rollout validation patterns. - The user needs reporting, export, compliance evidence, or operational proof. -## When not to use - -- The main problem is still choosing the high-level direction. -- The main problem is management-group, landing-zone, or subscription structure. -- The main problem is RBAC, managed identity, PIM, or Policy design. -- The task is a narrow implementation change with no operational design question. - ## Main domains covered - monitoring and observability posture @@ -72,17 +60,6 @@ For broader asks, return: - recovery or DR note when relevant - open operational risks -## Relationship to adjacent skills - -- `internal-azure-strategic` - Use first when the core decision is still unsettled. -- `internal-azure-organization-structure` - Use when the operations question is actually about management-group, subscription, or topology placement. -- `internal-azure-governance` - Use when the operations question is actually about RBAC, managed identity, Policy, or guardrail design rather than validation. -- `awesome-copilot-azure-resource-health-diagnose` - Use as depth support when the Azure resource is already identified and the next need is deep health diagnosis, log and telemetry analysis, or a remediation plan for that specific resource. - ## Common mistakes | Mistake | Why it matters | Instead | @@ -90,7 +67,7 @@ For broader asks, return: | Treating monitoring as proof that restore or recovery works | Healthy dashboards do not prove recovery viability | Keep monitoring evidence, backup proof, and restore proof as separate lines | | Skipping preflight for high-blast-radius rollout | Policy, identity, or connectivity failures surface too late | Define rollout unit, preflight checks, rollback trigger, and owner before rollout | | Reporting only control intent without operational evidence | The platform appears compliant without proof that it works | Record what Azure Monitor, Log Analytics, backup, or compliance signals actually showed | -| Mixing validation advice with new governance design instead of keeping the boundary clear | The operations skill stops being a reliable validation owner | Keep new Policy or RBAC design in `internal-azure-governance` and validate it here | +| Mixing validation advice with new governance design instead of keeping the boundary clear | The operations skill stops being a reliable validation owner | Keep new Policy or RBAC design out of the validation answer and validate the chosen design here | | Giving a DR answer without making the business criticality assumption visible | Recovery guidance can be overbuilt or incomplete | State the assumed criticality, RTO, or RPO before recommending the validation path | | Treating one successful rollout wave as proof for all subscriptions or regions | Wider inheritance, network, or residency paths can still fail differently | Validate the first safe unit and widen only after recording real evidence | diff --git a/.github/skills/internal-azure-organization-structure/SKILL.md b/.github/skills/internal-azure-organization-structure/SKILL.md index 0f11ee0a..f89d5355 100644 --- a/.github/skills/internal-azure-organization-structure/SKILL.md +++ b/.github/skills/internal-azure-organization-structure/SKILL.md @@ -1,20 +1,16 @@ --- name: internal-azure-organization-structure -description: Use when the user needs Azure control-plane or platform-structure guidance for tenant hierarchy, management groups, subscription models, landing-zone placement, environment segmentation, platform-level network topology, or other layout decisions that shape how Azure is organized before implementation. +description: Use when the user needs Azure tenant hierarchy, management-group layout, subscription models, landing-zone placement, environment segmentation, or platform-level network topology guidance before implementation. Do not use for RBAC, Policy, or identity design; monitoring, backup, or rollout validation; Azure DevOps pipelines; or materially ambiguous requests with no clear structural deliverable. --- # Internal Azure Organization Structure -## Referenced skills - -- `internal-azure-strategic`: route back when direction or tradeoff framing is still unsettled. -- `internal-azure-governance`: route RBAC, managed identity, PIM, Policy, tagging, or guardrail design. -- `internal-azure-operations`: route monitoring, validation, backup, Site Recovery, reporting, or audit evidence. - Use this skill when the next need is to design or review how Azure is structured at tenant and platform level. This skill owns Azure layout decisions, not generic strategy and not detailed RBAC or monitoring implementation. It helps translate a platform goal into management-group, subscription, landing-zone, topology, and rollout structure. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-azure`. + ## When to use - The user is shaping or reviewing Azure tenant hierarchy. @@ -23,13 +19,6 @@ This skill owns Azure layout decisions, not generic strategy and not detailed RB - The user needs platform-level network or regional structure guidance. - The user needs rollout-scope guidance for structural Azure change. -## When not to use - -- The question is mainly RBAC, PIM, managed identities, or Policy logic. -- The task is mainly monitoring, backup, reporting, or post-rollout validation. -- The user only needs generic strategic comparison with no concrete structure question. -- The task is already implementation-focused. - ## Main domains covered - tenant hierarchy @@ -73,25 +62,15 @@ For broader asks, return: - recommended placement model - smallest safe rollout unit - main risks -- what should move next to `internal-azure-governance` or `internal-azure-operations` - -## Relationship to adjacent skills - -- `internal-azure-strategic` - Use first when the user still needs broader decision framing or lens selection. -- `internal-azure-governance` - Use when the structural decision is accepted and the next need is RBAC, managed identity, PIM/PAM, Policy, or guardrail definition. -- `internal-azure-operations` - Use when the structure is accepted and the next need is validation, monitoring, backup, or operational evidence. ## Common mistakes | Mistake | Why it matters | Instead | | --- | --- | --- | | Proposing hierarchy or subscription layouts without a rollout scope | Tenant and hierarchy changes are hard to unwind if staged poorly | Name the smallest safe rollout unit: management group, subscription set, or region set | -| Mixing landing-zone placement and RBAC design into one vague answer | Structure and governance review get blurred together | Keep placement here and move authorization or guardrails to `internal-azure-governance` | +| Mixing landing-zone placement and RBAC design into one vague answer | Structure and governance review get blurred together | Keep placement here and keep authorization or guardrail design out of the structure answer | | Treating network topology as an operations concern instead of a structure concern | Connectivity design decisions get delayed until after layout is fixed | Keep hub-spoke, Virtual WAN, and regional topology in the structure lane | -| Using structure answers to sneak in Policy or RBAC design without separating the concerns | The ownership boundary becomes unreliable | State where the capability lives and hand off what controls or permissions apply | +| Using structure answers to sneak in Policy or RBAC design without separating the concerns | The ownership boundary becomes unreliable | State where the capability lives and keep what controls or permissions apply out of the structure answer | | Ignoring region or residency implications when they materially shape layout | Subscription or landing-zone placement can violate real requirements | Make sovereignty, region pairing, and continuity assumptions explicit | | Recommending platform subscriptions without naming their operating purpose | Platform estates become catch-all containers with unclear ownership | State whether the subscription is for connectivity, identity, management, or shared services | @@ -101,4 +80,4 @@ For broader asks, return: - Confirm the smallest safe rollout unit is named and matches the proposed structural change. - Confirm region or residency implications are explicit when they shape hierarchy, subscription, or topology choices. - Confirm platform ownership and workload ownership are separated when both appear in the recommendation. -- Confirm the next handoff is clear when the user now needs governance controls or operational proof. +- Confirm out-of-scope needs, such as governance controls or operational proof, are identified as outside this lane instead of being answered here. diff --git a/.github/skills/internal-azure-strategic/SKILL.md b/.github/skills/internal-azure-strategic/SKILL.md deleted file mode 100644 index 596a8dec..00000000 --- a/.github/skills/internal-azure-strategic/SKILL.md +++ /dev/null @@ -1,162 +0,0 @@ ---- -name: internal-azure-strategic -description: Use when the user needs high-level Azure platform decision support or tradeoff framing before implementation, and the next step is not yet structure, governance, operations, or code delivery. ---- - -# Internal Azure Strategic - -## Referenced skills - -- `awesome-copilot-azure-pricing`: pricing depth when Azure cost data can change the decision. -- `internal-azure-organization-structure`: route when tenant, management-group, subscription, or landing-zone layout becomes the decision. -- `internal-azure-governance`: route when RBAC, identity, Policy, tagging, or guardrail design becomes the decision. -- `internal-azure-operations`: route when validation, monitoring, backup, or audit evidence becomes the decision. -- `internal-terraform`: route implementation work in Terraform or OpenTofu. -- `internal-python-script`: route implementation work in standalone Python automation. -- `internal-bash-script`: route implementation work in standalone Bash automation. - -Use this skill when the main need is to reason about an Azure decision before implementation. - -This is a strategic support skill. It helps frame the decision, compare realistic options, expose tradeoffs, and recommend a direction. It does not implement the change and it does not choose Terraform, Python, or Bash on behalf of the user. - -## When to use - -- The user needs Azure decision support before execution. -- Multiple Azure approaches are credible and tradeoffs matter. -- The user wants a recommendation grounded in current Microsoft guidance. -- The user wants high-level support for platform, landing-zone, governance, resilience, cost, or operational decisions. - -## When not to use - -- The task is already a clear implementation change. -- The user only needs detailed RBAC, Policy, monitoring, backup, or automation implementation. -- The task is purely post-rollout validation or evidence gathering. -- The request is narrow and operational with no real decision to frame. - -## Main domains covered - -- platform and landing-zone decision framing -- Cloud Adoption Framework and Well-Architected guidance at decision level -- tenant, management-group, and subscription implications at decision level -- governance and identity implications at decision level -- operational implications at decision level -- resilience and continuity implications at decision level -- cost-value and FinOps implications at decision level - -## Optional lens activation - -Do not load every lens by default. - -Use only the minimum set of lenses needed for the request. If the user explicitly names one or more lenses, prioritize only those. If the user does not name lenses, infer the smallest useful set. - -Available lenses include: - -- security -- identity and access -- organization-structure -- governance -- operations -- monitoring and observability -- BC/DR -- FinOps -- compliance -- rollout and rollback -- blast radius -- maintainability - -Rules: - -- Start narrow. -- Expand only when the request is broad, risky, or ambiguous. -- If another lens would materially improve the recommendation, suggest it briefly instead of forcing it. -- Keep the active lenses explicit when more than one is in play. - -Load `references/lens-playbook.md` when the user wants a deeper framing aid or when the choice of lenses is not obvious. - -## Optional BC/DR lens - -BC/DR is optional. - -Activate it only when: - -- the user asks about resilience, backup, recovery, failover, RTO, RPO, or regional continuity -- the decision has clear continuity implications -- the recommendation would be materially incomplete without it - -If BC/DR seems relevant but is not requested, suggest it as an optional lens instead of forcing it. - -## Use of current documentation - -Use current Microsoft documentation only when freshness materially affects the answer, especially for Azure service support, landing-zone guidance updates, Policy behavior, RBAC semantics, regional capability, or service limits. - -Do not invoke current-doc research by default for stable, generic reasoning. - -## Mandatory behavior - -- Identify the decision first, not the implementation tool. -- Make assumptions explicit. -- Compare realistic options, not strawmen. -- Keep tradeoffs concrete. -- Surface material risk, blast radius, and reversibility when relevant. -- Include cost-value considerations when they matter to the decision. -- Stay proportional to the size of the question. - -## Adaptive output modes - -Choose the lightest output that fits the request. - -### Quick answer - -Use for narrow asks. - -Include: - -- direct recommendation -- short rationale -- optional risk or follow-up note - -### Decision note - -Use for normal strategic support. - -Include: - -- decision statement -- key options or tradeoff -- recommended direction -- main risk or validation note - -### Deep analysis - -Use only for broad, ambiguous, high-risk, or explicitly detailed requests. - -Include: - -- context and assumptions -- options considered -- active lenses used -- recommendation and why it wins -- main risks and blast radius -- validation or follow-up path - -## Relationship to adjacent skills - -- `awesome-copilot-azure-pricing` - Use as depth support when the strategic question becomes SKU pricing, spend estimation, or Azure-specific cost comparison that needs current pricing data. -- `internal-azure-organization-structure` - Use when the next need is tenant hierarchy, management-group layout, subscription model, landing-zone placement, or platform topology. -- `internal-azure-governance` - Use when the next need is RBAC model, managed identity boundaries, PIM/PAM posture, Policy design, or guardrail definition. -- `internal-azure-operations` - Use when the next need is Azure Monitor, backup, Site Recovery validation, reporting, audit evidence, preflight, or post-rollout checks. -- `internal-terraform`, `internal-python-script`, `internal-bash-script` - Use when the decision is settled and implementation begins. - -## Anti-patterns - -- forcing a full multi-lens analysis for a small question -- treating BC/DR as mandatory for every answer -- recommending a direction without current-source verification when freshness matters -- confusing decision support with implementation guidance -- expanding into tool selection when the user did not ask for it -- giving generic best-practice advice without context, tradeoff, or cost implication diff --git a/.github/skills/internal-azure-strategic/agents/openai.yaml b/.github/skills/internal-azure-strategic/agents/openai.yaml deleted file mode 100644 index 091ed2b6..00000000 --- a/.github/skills/internal-azure-strategic/agents/openai.yaml +++ /dev/null @@ -1,4 +0,0 @@ -interface: - display_name: "Internal Azure Strategic" - short_description: "Azure decision framing and tradeoff support" - default_prompt: "Use $internal-azure-strategic to frame this Azure decision, keep the answer proportional, and activate only the lenses that matter." diff --git a/.github/skills/internal-azure/SKILL.md b/.github/skills/internal-azure/SKILL.md new file mode 100644 index 00000000..cd8579b3 --- /dev/null +++ b/.github/skills/internal-azure/SKILL.md @@ -0,0 +1,165 @@ +--- +name: internal-azure +description: Use when an Azure task cannot be routed confidently to a specific Azure skill because the request is materially ambiguous, has multiple Azure domains with no clear primary owner, or requires clarification before selecting the correct specialist, or when the user needs high-level Azure platform decision support or tradeoff framing before implementation. Do not use for clearly scoped organization structure, governance or identity, operations or validation, or Azure DevOps pipeline tasks. +--- + +# Internal Azure + +## Referenced skills + +- `internal-azure-organization-structure`: tenant, management-group, subscription, landing-zone, and platform-topology owner. +- `internal-azure-governance`: RBAC, managed identity, PIM, Policy, tagging, and guardrail owner. +- `internal-azure-operations`: monitoring, validation, backup, Site Recovery, reporting, and evidence owner. +- `internal-azure-devops`: Azure DevOps pipeline and project-automation owner. +- `awesome-copilot-azure-pricing`: Azure-specific pricing depth when cost data is the primary problem. + +Fallback router for Azure tasks that cannot be assigned confidently to one specialist, and strategic support skill for high-level Azure decision framing. Do not activate only because the task concerns Azure; activate only when material routing uncertainty blocks owner selection or when the user needs decision support before the next step is structure, governance, operations, or delivery. Do not activate when one specialist clearly owns the next step. + +## When to use + +- Use this fallback when material ambiguity prevents selecting one primary Azure specialist. +- Use it when multiple Azure domains are material and no primary owner can be identified safely. +- Use it when the user explicitly invokes `$internal-azure`. +- Use it when the user needs high-level Azure decision support or tradeoff framing before implementation. + +## When not to use + +- The task is already a clear implementation change. +- The user only needs detailed RBAC, Policy, monitoring, backup, or pipeline implementation detail. +- The task is purely post-rollout validation or evidence gathering. +- The request is narrow and operational with no real decision to frame. + +## Routing threshold + +Activate only when at least one condition holds: + +- the request is materially ambiguous and clarification is required before an Azure owner can be selected; +- multiple Azure domains are material and no primary owner can be identified safely; +- the task asks which Azure problem-solving lane should own the work before requesting a domain solution; +- the user needs strategic decision framing and the next step is not yet structure, governance, operations, or delivery. + +Explicit `$internal-azure` invocation remains valid. + +## Dispatch contract + +1. State the routing uncertainty. +2. Identify the candidate Azure owners. +3. Select the minimum specialist set. +4. Keep strategic comparison here only while it is needed to choose the owner. +5. Hand the resolved task to the primary specialist instead of retaining ownership. + +## Optional lens activation + +Do not load every lens by default. + +Use only the minimum set of lenses needed for the request. If the user explicitly names one or more lenses, prioritize only those. If the user does not name lenses, infer the smallest useful set. + +Available lenses include: + +- security +- identity and access +- organization-structure +- governance +- operations +- monitoring and observability +- BC/DR +- FinOps +- compliance +- rollout and rollback +- blast radius +- maintainability + +Rules: + +- Start narrow. +- Expand only when the request is broad, risky, or ambiguous. +- If another lens would materially improve the recommendation, suggest it briefly instead of forcing it. +- Keep the active lenses explicit when more than one is in play. + +Load `references/lens-playbook.md` when the user wants a deeper framing aid or when the choice of lenses is not obvious. + +## Optional BC/DR lens + +BC/DR is optional. + +Activate it only when: + +- the user asks about resilience, backup, recovery, failover, RTO, RPO, Site Recovery, or regional continuity +- the decision has clear continuity implications +- the recommendation would be materially incomplete without it + +If BC/DR seems relevant but is not requested, suggest it as an optional lens instead of forcing it. + +## Use of current documentation + +Use current Microsoft documentation only when freshness materially affects the answer, especially for Azure service support, landing-zone guidance updates, Policy behavior, RBAC semantics, regional capability, or service limits. + +Do not invoke current-doc research by default for stable, generic reasoning. + +## Mandatory behavior + +- Identify the decision first, not the implementation tool. +- Make assumptions explicit. +- Compare realistic options, not strawmen. +- Keep tradeoffs concrete. +- Surface material risk, blast radius, and reversibility when relevant. +- Include cost-value considerations when they matter to the decision. +- Stay proportional to the size of the question. + +## Adaptive output modes + +Choose the lightest output that fits the request. + +### Quick answer + +Use for narrow asks. + +Include: + +- direct recommendation +- short rationale +- optional risk or follow-up note + +### Decision note + +Use for normal strategic support. + +Include: + +- decision statement +- key options or tradeoff +- recommended direction +- main risk or validation note + +### Deep analysis + +Use only for broad, ambiguous, high-risk, or explicitly detailed requests. + +Include: + +- context and assumptions +- options considered +- active lenses used +- recommendation and why it wins +- main risks and blast radius +- validation or follow-up path + +## Anti-patterns + +- forcing a full multi-lens analysis for a small question +- treating BC/DR as mandatory for every answer +- recommending a direction without current-source verification when freshness matters +- activating this fallback when one specialist clearly owns the next step +- expanding into tool selection when the user did not ask for it +- giving generic best-practice advice without context, tradeoff, or cost implication + +## Validation + +- State why the request could not be assigned to one primary Azure specialist, or name the decision being framed. +- Confirm the selected specialist set is the minimum needed to resolve the uncertainty. +- Confirm the decision statement is explicit and narrow enough that the next owner is obvious. +- Confirm assumptions, active lenses, and the main tradeoff are named instead of implied. +- Confirm the recommendation includes reversibility or blast-radius guidance when the choice is hard to unwind. +- Confirm cost-value or operational impact is called out when it materially changes the recommendation. +- Confirm the answer states when freshness matters and which current Microsoft fact still needs validation. +- Confirm the resolved task is handed to a primary specialist. diff --git a/.github/skills/internal-azure/agents/openai.yaml b/.github/skills/internal-azure/agents/openai.yaml new file mode 100644 index 00000000..1f7ce6f5 --- /dev/null +++ b/.github/skills/internal-azure/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "Internal Azure" + short_description: "Azure routing and strategic decision support" + default_prompt: "Use $internal-azure to route an unclear Azure task to the minimum specialist set, or to frame an Azure decision when the next step is not yet structure, governance, operations, or delivery." diff --git a/.github/skills/internal-azure-strategic/references/lens-playbook.md b/.github/skills/internal-azure/references/lens-playbook.md similarity index 53% rename from .github/skills/internal-azure-strategic/references/lens-playbook.md rename to .github/skills/internal-azure/references/lens-playbook.md index f355b37d..d39c79ab 100644 --- a/.github/skills/internal-azure-strategic/references/lens-playbook.md +++ b/.github/skills/internal-azure/references/lens-playbook.md @@ -20,6 +20,26 @@ Use this reference when the user wants more depth than the base skill should loa - The rollout adds monitoring, backup, or validation burden: suggest `operations` - A failure would interrupt critical platform capability: suggest `BC/DR` +## Decision note pattern + +Use this when the question is too consequential for a quick answer but does not need a full deep analysis. + +1. Decision statement: what Azure choice is being made. +2. Assumptions: what current state, constraints, or timelines the recommendation depends on. +3. Viable options: usually two or three realistic Azure-local paths. +4. Recommendation: which option wins and why. +5. Tradeoffs and blast radius: what gets better, what gets harder, and what is hard to reverse. +6. Validation note: what current-fact check, proof, or next-owner handoff is still required. + +## When to stay quick answer versus upgrade to a decision note + +| Stay in `Quick answer` when | Upgrade to `Decision note` when | +| --- | --- | +| One option is clearly better and the downside is local | At least two Azure-local options are still viable | +| The choice does not alter the platform, identity, or recovery posture | The choice changes management groups, subscriptions, delegated access, or continuity expectations | +| The answer can stay within one lens without hiding material risk | A second lens changes the recommendation or the risk statement | +| Freshness is not the deciding factor | Current Azure behavior, service support, or limits could change the outcome | + ## Depth control - Stay in `Quick answer` mode when one option is clearly better and the user asked a narrow question. diff --git a/.github/skills/internal-azure/references/routing-matrix.md b/.github/skills/internal-azure/references/routing-matrix.md new file mode 100644 index 00000000..499b5cc0 --- /dev/null +++ b/.github/skills/internal-azure/references/routing-matrix.md @@ -0,0 +1,30 @@ +# Azure Routing Scenario Matrix + +## Fallback-positive cases + +| Scenario | Why no primary owner | +|---|---| +| Underspecified cross-subscription control problem mixing structure, governance, and operations with no clear primary deliverable | The request names multiple Azure domains but does not identify which deliverable takes priority. | +| Azure platform question asking which lane should own the work without naming structure, governance, operations, or pipelines | The user has not selected a domain; the fallback must clarify the lane before any specialist can engage. | +| Broad Azure adoption review where the user wants a general health assessment across all domains | No single specialist owns a cross-domain health review; the fallback selects the minimum set. | + +## Direct-specialist negative cases + +| Scenario | Direct owner | Reason | +|---|---|---| +| Management-group or subscription layout | `internal-azure-organization-structure` | The deliverable is a structural placement decision. | +| RBAC, Policy, or PIM design | `internal-azure-governance` | The deliverable is a guardrail or identity boundary. | +| Backup/restore or DR validation | `internal-azure-operations` | The deliverable is operational evidence or continuity proof. | +| Pipeline YAML or project automation | `internal-azure-devops` | The deliverable is pipeline behavior or project flow. | + +## Multi-domain primary-owner cases + +| Scenario | Primary owner | Secondary | Reason | +|---|---|---|---| +| Subscription design with later Policy work | `internal-azure-organization-structure` | `internal-azure-governance` | The first deliverable is placement; Policy follows once the structure is settled. | +| Policy rollout evidence | `internal-azure-governance` | `internal-azure-operations` | The first deliverable is governance design; operations validates the rollout. | +| Pipeline permissions detail | `internal-azure-devops` or `internal-azure-governance` | depends on deliverable | Choose `internal-azure-devops` when the deliverable is pipeline behavior; choose `internal-azure-governance` when the deliverable is the permission boundary. | + +## Review rule + +Prefer a direct specialist whenever a reasonable reviewer can name one primary owner from the request itself. Activate the fallback only when the request does not identify a primary owner and clarification is required before a specialist can engage. diff --git a/.github/skills/internal-copilot-audit/SKILL.md b/.github/skills/internal-copilot-audit/SKILL.md index 2913cb49..c60c3cc0 100644 --- a/.github/skills/internal-copilot-audit/SKILL.md +++ b/.github/skills/internal-copilot-audit/SKILL.md @@ -36,7 +36,7 @@ For skill bundles, treat `references/`, `scripts/`, `assets/`, and `agents/opena - Detect sync workflows that skip or fail to report governance review for `.github/copilot-instructions.md` and root `AGENTS.md`. - Detect naming violations and stale inventory references. - Detect governance files that still describe removed, renamed, or retired assets. -- Detect catalog retirements or remaps that were not propagated in the same change to the supported consistency and sync entrypoints, `./.github/scripts/run.sh check_catalog_consistency` and `./.github/scripts/run.sh sync_copilot_catalog`. +- Detect catalog retirements or remaps that were not propagated in the same change to the supported consistency entrypoint `./.github/scripts/run.sh check_catalog_consistency`. - Detect token-risk claims that rely on assumed runtime loading behavior instead of observable repository signals such as description length, exact trigger collisions, or repo-profile coverage. ## Audit Order diff --git a/.github/skills/internal-excel/SKILL.md b/.github/skills/internal-excel/SKILL.md index 60efe124..fe7e9064 100644 --- a/.github/skills/internal-excel/SKILL.md +++ b/.github/skills/internal-excel/SKILL.md @@ -7,7 +7,7 @@ description: Use when tasks involve CSV, TSV, or Excel tabular data profiling, s ## Referenced skills -- `openai-spreadsheet`: on-demand support owner when rendered fidelity, cached recalculation, charts, or polished workbook presentation are the primary requirement. +- `anthropic-xlsx`: on-demand support owner when rendered fidelity, cached recalculation, charts, or polished workbook presentation are the primary requirement. ## When to use @@ -19,7 +19,7 @@ description: Use when tasks involve CSV, TSV, or Excel tabular data profiling, s ## When not to use -- Charts, rendered review, cached recalculation, or polished workbook presentation as a first-class delivery requirement; use `openai-spreadsheet`. +- Charts, rendered review, cached recalculation, or polished workbook presentation as a first-class delivery requirement; use `anthropic-xlsx`. - Single-language implementation work after the tabular-data approach is already chosen; use the narrower file or runtime owner for that code. - Database or warehouse design work that is not primarily about local CSV, TSV, or Excel artifacts. @@ -27,7 +27,7 @@ description: Use when tasks involve CSV, TSV, or Excel tabular data profiling, s - This skill owns data integrity first: headers, types, nulls, duplicates, joins, reconciliation, stable IDs, delimiter detection, encoding, locale-sensitive numeric fields, and row-count preservation. - Keep evidence compact for large files: report headers, counts, targeted anomalies, transformation rules, exact validation gaps, and any locale or coercion assumptions that change numeric meaning. -- Treat `.xlsx` as a workbook container first. Stay here for tabular extraction, safe value-level transformation, workbook contract inspection, writer parity checks, and formula-column decisions. Route to `openai-spreadsheet` only when rendered fidelity, cached recalculation, charts, or presentation-preserving delivery is the main job. +- Treat `.xlsx` as a workbook container first. Stay here for tabular extraction, safe value-level transformation, workbook contract inspection, writer parity checks, and formula-column decisions. Route to `anthropic-xlsx` only when rendered fidelity, cached recalculation, charts, or presentation-preserving delivery is the main job. - Preserve identifier fidelity before coercion: keep leading zeros, long numeric IDs, and day-first date text exact until an explicit schema rule says otherwise. - Flag spreadsheet-bound text that could execute as a formula and minimize sensitive-column exposure in samples or logs. - Read `references/tool-selection.md` when choosing between Python `csv`, `pandas`, `openpyxl`, PyArrow, Polars, or DuckDB, or when scale, memory use, file format, or workbook fidelity makes the engine choice non-obvious. @@ -52,7 +52,7 @@ description: Use when tasks involve CSV, TSV, or Excel tabular data profiling, s - When a sample workbook or sample tab defines the expected layout, treat that tab as a contract: column order, widths, header style, body style, money style, alert or error row style, and formula columns. - Verify that every tab produced by the same writer receives the same contract where applicable, not just the sampled tab. - When the user asks for verifiable formulas, keep derived columns as Excel formulas and keep source columns as atomic input values instead of precomputed results. -- If workbook behavior stays contract-level and formula-level, keep the guidance here. Route to `openai-spreadsheet` only when visual polish, charts, or recalculated delivery artifacts become primary. +- If workbook behavior stays contract-level and formula-level, keep the guidance here. Route to `anthropic-xlsx` only when visual polish, charts, or recalculated delivery artifacts become primary. ## Binary Artifacts diff --git a/.github/skills/internal-excel/references/tool-selection.md b/.github/skills/internal-excel/references/tool-selection.md index d0a87656..2117e176 100644 --- a/.github/skills/internal-excel/references/tool-selection.md +++ b/.github/skills/internal-excel/references/tool-selection.md @@ -12,7 +12,7 @@ Read this reference when scale, file format, or workbook fidelity makes the engi | Columnar scans, type-stable interchange, Parquet or Arrow conversion | PyArrow | Strong schema control and efficient IO | Prefer when conversion fidelity and explicit types matter. | | Large in-memory transforms with strict schema and speed focus | Polars | Fast columnar execution | Prefer when `pandas` becomes memory-heavy or slow. | | Reading or updating `.xlsx` cell values while workbook UX is secondary | `openpyxl` | Native workbook access | Stay here only when layout, styles, and rendered review are not the main goal. | -| Formulas, styles, charts, cached recalculation, rendered review, workbook preservation | `openai-spreadsheet` | Workbook UX owner | Route out of this skill. | +| Formulas, styles, charts, cached recalculation, rendered review, workbook preservation | `anthropic-xlsx` | Workbook UX owner | Route out of this skill. | ## Selection rules @@ -22,9 +22,9 @@ Read this reference when scale, file format, or workbook fidelity makes the engi - Prefer DuckDB, Polars, or PyArrow when file size or join volume makes full `pandas` loads risky. - Prefer engines that support explicit schema declarations when inference could corrupt IDs, dates, currency, or nullable numeric fields. -## Escalation to `openai-spreadsheet` +## Escalation to `anthropic-xlsx` -Route to `openai-spreadsheet` when any of these are first-class requirements: +Route to `anthropic-xlsx` when any of these are first-class requirements: - formulas or cached recalculation - charts or workbook layout diff --git a/.github/skills/internal-gateway-codebase-improvement/SKILL.md b/.github/skills/internal-gateway-codebase-improvement/SKILL.md new file mode 100644 index 00000000..0d13cf10 --- /dev/null +++ b/.github/skills/internal-gateway-codebase-improvement/SKILL.md @@ -0,0 +1,82 @@ +--- +name: internal-gateway-codebase-improvement +description: Use when manually improving codebase architecture and implementation clarity. +disable-model-invocation: true +--- + +# Internal Gateway Codebase Improvement + +## Referenced skills + +- `/mattpocock-improve-codebase-architecture`: architecture discovery and + candidate report owner. +- `/addyosmani-code-simplification`: behavior-preserving implementation + simplification owner. +- `/internal-tdd`: executable or evaluable behavior-change gate. +- `/superpowers-verification-before-completion`: final evidence owner. + +## Manual invocation boundary + +This skill runs only after the user invokes it explicitly. It is not a +canonical gateway, implicit fallback, peer-dispatch target, or subagent target. + +## When to use + +- The user explicitly requests codebase improvement and the evidence supports + one of the three supported lanes. + +## When not to use + +- Feature development, performance optimization, security remediation, or + dependency upgrades. +- Work that needs a canonical gateway or automatic routing. + +## Lane selection + +Select exactly one lane from repository evidence: + +- `local-simplification`: readability, naming, nesting, duplication, dead code, + or unnecessary implementation abstraction inside an already valid boundary. +- `architecture-improvement`: shallow modules, leaking seams, poor locality, + cross-module coupling, or testability constrained by current interfaces. +- `combined`: an approved architecture refactor whose changed implementation + also contains bounded simplification opportunities. + +Do not run both source methods by default. No silent lane escalation: stop and +ask before changing from `local-simplification` to an architecture lane. + +## Core workflow + +1. Recover explicit target and anti-scope. +2. Inspect bounded evidence and choose one lane. +3. State the lane, reason, expected files, and validation path. +4. Establish a Passing behavior baseline. +5. For architecture lanes, run architecture discovery and stop at the + Structural Approval Gate before any write. +6. For executable changes, load `/internal-tdd`. +7. Record the approved interfaces and seams as the Protected seam set. +8. Apply the executable refactor in the approved scope. +9. Apply behavior-preserving simplification only for + `local-simplification` or the approved changed scope of `combined`. +10. Run the focused checks and the Final Evidence Gate. + +Domain-model and ADR writes from the architecture method also require the +Structural Approval Gate. + +## Structural Approval Gate + +Before any architecture, domain-model, or ADR write, present the candidate +report and the expected file set. Stop and wait for explicit user approval. +Do not proceed without it. + +## Protected seam set + +Before any executable refactor, record the approved modules, interfaces, +adapters, side effects, error behavior, ordering, and test surfaces. +Simplification must not alter any protected seam. + +## Final Evidence Gate + +After all writes, load `/superpowers-verification-before-completion` and +present fresh passing evidence for the focused validation path before +claiming completion. diff --git a/.github/skills/internal-gateway-codebase-improvement/agents/openai.yaml b/.github/skills/internal-gateway-codebase-improvement/agents/openai.yaml new file mode 100644 index 00000000..4ed96624 --- /dev/null +++ b/.github/skills/internal-gateway-codebase-improvement/agents/openai.yaml @@ -0,0 +1,6 @@ +interface: + display_name: "Codebase Improvement" + short_description: "Manually improve architecture and code clarity" + default_prompt: "Use /internal-gateway-codebase-improvement to manually improve codebase architecture and implementation clarity. Select exactly one lane from repository evidence, establish a passing behavior baseline, protect approved seams, and run the final evidence gate before claiming completion." +policy: + allow_implicit_invocation: false diff --git a/.github/skills/internal-gateway-codebase-improvement/references/workflow.md b/.github/skills/internal-gateway-codebase-improvement/references/workflow.md new file mode 100644 index 00000000..c086a680 --- /dev/null +++ b/.github/skills/internal-gateway-codebase-improvement/references/workflow.md @@ -0,0 +1,107 @@ +# Codebase Improvement Workflow + +## State machine + +```mermaid +flowchart TD + A[Manual user invocation] --> B[Bounded evidence] + B --> C{Select exactly one lane} + C -->|local-simplification| D[Passing behavior baseline] + C -->|architecture-improvement| E[Architecture candidate report] + C -->|combined| E + E --> F[Structural Approval Gate] + F -->|rejected| X[Stop without writes] + F -->|approved| G[Passing behavior baseline] + G --> H[Protected seam set] + H --> I[Executable refactor] + D --> J[Behavior-preserving simplification] + I --> K{Combined lane?} + K -->|no| L[Focused validation] + K -->|yes| J + J --> L + L --> M[Final Evidence Gate] +``` + +## Lane selection signals + +### `local-simplification` + +Signals: readability, naming, nesting, duplication, dead code, or unnecessary +implementation abstraction inside an already valid boundary. The module +structure, interfaces, and side-effect shape are not in question. + +Anti-signals: the change requires new modules, new interfaces, cross-module +coupling changes, or testability improvements that depend on interface +redesign. + +### `architecture-improvement` + +Signals: shallow modules, leaking seams, poor locality, cross-module coupling, +or testability constrained by current interfaces. The improvement requires +changing approved module boundaries or interface contracts. + +Anti-signals: the problem is confined to implementation clarity within an +already valid module boundary and does not require structural change. + +### `combined` + +Signals: an approved architecture refactor whose changed implementation also +contains bounded simplification opportunities. Both structural and local +clarity improvements are evidenced. + +Anti-signals: either the structural or the simplification case is not +supported by bounded evidence. + +## No silent lane escalation + +If the initial evidence selects `local-simplification` but the work reveals +an architecture problem, stop and ask the user before changing lanes. Do not +proceed into architecture writes on a `local-simplification` selection. + +## Structural Approval Gate + +For `architecture-improvement` and `combined` lanes, present the candidate +report, expected file set, and affected interfaces before any write. Stop and +wait for explicit user approval. Domain-model and ADR writes require the same +approval boundary. + +## Passing behavior baseline + +Before any executable refactor, establish that the current code passes its +focused validation. Record the command and result. This baseline is the +reference point for the post-refactor check. + +## Protected seam set + +Before any executable refactor, record the approved modules, interfaces, +adapters, side effects, error behavior, ordering, and test surfaces that the +refactor must not alter. Simplification applies only within these bounds. + +## Behavior-preserving simplification + +Apply `/addyosmani-code-simplification` only for `local-simplification` or the +approved changed scope of `combined`. The simplification must not alter any +entry in the Protected seam set. + +## Executable refactor + +For executable changes, load `/internal-tdd` before implementation. Apply the +refactor within the approved scope and protected seam set. + +## Focused validation + +Run the focused test or check identified during lane selection. The check +must cover the changed files and their direct dependents. + +## Final Evidence Gate + +Load `/superpowers-verification-before-completion` and present fresh passing +evidence before claiming completion. + +## Stop conditions + +- Missing Passing behavior baseline. +- Unclear ownership or affected-file set. +- Focused validation fails after refactor. +- Simplification would change a protected architecture seam. +- Evidence does not support the selected lane. diff --git a/.github/skills/internal-gateway-critical-master/SKILL.md b/.github/skills/internal-gateway-critical-master/SKILL.md index fb569a2e..ef6f0145 100644 --- a/.github/skills/internal-gateway-critical-master/SKILL.md +++ b/.github/skills/internal-gateway-critical-master/SKILL.md @@ -32,17 +32,16 @@ Run exactly three phases. Do not skip a phase and do not loop back unless new ev - Read only the smallest evidence needed to understand the proposal, decision, or assumption set. - Identify the material claims, constraints, success criteria, and anti-scope. -- Output: a one-paragraph summary of what is being challenged and why it matters now. +- Record internally: what is being challenged and why it matters now. ### Phase 2: Challenge -- Select **2-3 lenses** from the table below based on the highest-risk gaps in the summary. -- The **third lens must be lateral**: `analogy` or `reverse assumption`. +- Select exactly **three lenses** from the table below based on the highest-risk gaps in the summary. +- Lens three must be lateral: `analogy` or `reverse-assumption`. - Apply one optional pre-mortem pass if failure modes are material and not covered by the selected lenses. - Lead with the strongest supported objection first. Stop at one finding when that objection controls the decision; do not pad findings. -- Ask exactly one concise root question only when the answer would materially change the critique, and put it in `finding.question`. +- Ask at most one concise root question across all findings when the answer would materially change the critique. - Treat mitigations as conditions to continue, not as implementation designs that rescue the proposal. -- Output: 1-3 raw findings, each with a claim class and a note on evidence quality. | Lens | Question | Use when | | --- | --- | --- | @@ -64,22 +63,31 @@ Trigger a pre-mortem when at least one of these is true: - The plan introduces a new operational owner, on-call rotation, or handoff. - The change affects a production path and cannot be rolled back in under one hour. -For a pre-mortem, state one concrete failure, list the 2-3 most likely root causes with classification and probability, and define a required mitigation for each `high` or `medium` cause. - ### Phase 3: Synthesize - Run the Final Consistency Gate: name the strongest supported objection, downgrade weak claims to hypotheses, and surface unresolved uncertainty. -- When the user has already defended the proposal, classify the defense as `resolves`, `narrows`, `accepts-risk`, or `unanswered`, then name the strongest defense and the remaining vulnerability inside the synthesis. -- Format the result using the contract in `references/output-contract.md`. -- Recommend exactly one outcome from `## Outcome meanings`. +- Set Defense to one of `none`, `resolves`, `narrows`, `accepts-risk`, or `unanswered`. +- When Defense is not `none`, name the strongest defense and the remaining vulnerability inside the synthesis. +- Select exactly one canonical routing outcome from `## Outcome meanings`. + +## Internal critical record + +Keep the following as internal working state. Do not print the internal critical record in normal chat; use it to produce the public card. + +- Challenged proposal and timing +- Selected lenses (exactly three; third is lateral) +- Material claims with claim class (`confirmed`, `inference`, `estimate`) and evidence quality (`strong`, `partial`, `weak`) +- Strongest objection +- Defense classification and, when not `none`, strongest defense and remaining vulnerability +- Pre-mortem failure, causes, and mitigations when triggered +- Unresolved uncertainty +- Exactly one canonical routing outcome + +Material risk and decisive uncertainty must never be hidden. Details are available when the user asks. The visible labels match the user's language. The critique line states both what is wrong and one concrete reason. The old multi-section report is forbidden. -## Token Budget +## Public projection -- Target output: **600 words or fewer** per challenge cycle. -- Maximum findings: **3**. -- Per-field limits are authoritative; see `references/output-contract.md`. -- Maximum synthesis: **100 words**. -- If the topic demands more depth, split the work into another critical cycle. +In normal chat, emit only the localized three-to-five-line emoji card defined in `references/output-contract.md`. ## Claim Discipline @@ -90,11 +98,12 @@ For a pre-mortem, state one concrete failure, list the 2-3 most likely root caus ## Tooling -- Optional: `scripts/validate_critical_output.py` checks a rendered output against the contract in `references/output-contract.md`. +- Optional: `scripts/validate_critical_output.py` checks a rendered card against the contract in `references/output-contract.md`. - The optional validator and its pure helper live inside this skill bundle so the skill can be copied without depending on repo-global Python modules. - Reuse `fixtures/critical_output_valid.md` and sibling fixture samples instead of repeating long inline payloads. - Follow `references/maintenance-guidance.md` for fixture reuse and cache-aware search discipline. - Keep this bundle self-contained: do not require instructions, examples, or enforcement rules from outside this directory. +- Script output contract: `text` for short operator summaries (default), `json` for nested or machine-consumed output, `compact` for status and counts; validation findings on stdout, file and usage failures on stderr; keep output bounded. ## Outcome meanings @@ -102,7 +111,7 @@ For a pre-mortem, state one concrete failure, list the 2-3 most likely root caus | --- | --- | | `reformulate-plan` | Planning must be rewritten. | | `de-escalate-to-simple` | A concrete local task remains. | -| `execute-clear-next-step` | Execution is approved and clear. | +| `route-to-execution-owner` | The plan is challenge-ready and an execution owner can proceed; this is routing readiness, not execution approval. | | `review-evidence` | The next risk is correctness evidence. | -| `continue-critical` | Another pressure-test loop is needed. | +| `continue-critical-with-new-evidence` | Another pressure-test loop is needed; legal only when the synthesis names the new evidence required for the next cycle. | | `accept-with-risk` | The user may proceed while accepting a named residual risk. | diff --git a/.github/skills/internal-gateway-critical-master/agents/openai.yaml b/.github/skills/internal-gateway-critical-master/agents/openai.yaml index 7d72cc37..721779a2 100644 --- a/.github/skills/internal-gateway-critical-master/agents/openai.yaml +++ b/.github/skills/internal-gateway-critical-master/agents/openai.yaml @@ -1,4 +1,4 @@ interface: display_name: "Internal Gateway Critical Master" short_description: "Critical challenge for plans and decisions" - default_prompt: "Use $internal-gateway-critical-master to pressure-test this plan or decision. Follow the three-phase procedure in SKILL.md, select 2-3 lenses (with the third being lateral: analogy or reverse assumption), and close with the structured output contract in references/output-contract.md. Keep the total response under 600 words." + default_prompt: "Use /internal-gateway-critical-master to pressure-test this plan or decision. Complete the three internal phases and retain the critical record internally. In normal chat, emit only the localized three-to-five-line emoji card from references/output-contract.md, in the user's language. Always explain the critique with one concrete reason, surface any material risk or decision-changing question, and keep canonical classifications and routing codes internal unless the user asks for details." diff --git a/.github/skills/internal-gateway-critical-master/fixtures/critical_output_advisory.md b/.github/skills/internal-gateway-critical-master/fixtures/critical_output_advisory.md deleted file mode 100644 index fc46bd71..00000000 --- a/.github/skills/internal-gateway-critical-master/fixtures/critical_output_advisory.md +++ /dev/null @@ -1,19 +0,0 @@ -## Summary - -This summary intentionally exceeds the allowed word budget so the validator emits a non-blocking finding while the overall structure still remains valid for strict-mode coverage. It does so by repeating the same narrow point in several different clauses, which is exactly the kind of padding the contract is supposed to reject. The content still looks structurally correct, but the summary itself should cross the seventy-five word threshold and leave the rest of the document usable for strict-mode regression coverage. - -## Findings - -### 1. The output is still structurally valid - -- **Impact:** The content can be reviewed, but it should still trip strict mode because the summary is too long. -- **Evidence:** `inference` - the section uses a valid finding shape. -- **Mitigation:** Shorten the summary back under the limit. - -## Synthesis - -The output is structurally valid, but the word-budget finding should cause `--strict` to fail. - -## Outcome - -`accept-with-risk` diff --git a/.github/skills/internal-gateway-critical-master/fixtures/critical_output_invalid_missing_section.md b/.github/skills/internal-gateway-critical-master/fixtures/critical_output_invalid_missing_section.md deleted file mode 100644 index 694cfdca..00000000 --- a/.github/skills/internal-gateway-critical-master/fixtures/critical_output_invalid_missing_section.md +++ /dev/null @@ -1,15 +0,0 @@ -## Summary - -This output is missing the required outcome section. - -## Findings - -### 1. The output omits a required contract field - -- **Impact:** The validator cannot confirm the final decision. -- **Evidence:** `confirmed` - the `## Outcome` section is absent. -- **Mitigation:** Add the missing `## Outcome` section and wrap the value in backticks. - -## Synthesis - -The structure is incomplete, so the output should fail validation. diff --git a/.github/skills/internal-gateway-critical-master/fixtures/critical_output_valid.md b/.github/skills/internal-gateway-critical-master/fixtures/critical_output_valid.md index c6cd0a9b..3b7d0641 100644 --- a/.github/skills/internal-gateway-critical-master/fixtures/critical_output_valid.md +++ b/.github/skills/internal-gateway-critical-master/fixtures/critical_output_valid.md @@ -1,21 +1,3 @@ -## Summary - -We are challenging a proposal to move validation logic from CI into a pre-commit hook. The change matters now because it affects every contributor's workflow and could hide failures from the central audit log. - -## Findings - -### 1. The audit trail weakens - -- **Impact:** Central CI logs become incomplete for compliance reviews. -- **Evidence:** `inference` — no replacement logging is described. -- **Mitigation:** Add a signed attestation step before the hook is enabled. -- **Reframe:** Treat local validation as an early filter, not a replacement for CI. -- **Question:** Which central audit record replaces the CI validation log? - -## Synthesis - -The strongest risk is compliance visibility, not implementation effort. The proposal can work if the mitigation is accepted. - -## Outcome - -`accept-with-risk` +🎯 **Plan:** Move validation from CI to developer machines. +⚠️ **Critique:** Central proof disappears because local checks do not create a shared record. +✅ **Advice:** Keep CI until an equivalent central control exists. diff --git a/.github/skills/internal-gateway-critical-master/fixtures/critical_output_valid_premortem.md b/.github/skills/internal-gateway-critical-master/fixtures/critical_output_valid_premortem.md new file mode 100644 index 00000000..93a46988 --- /dev/null +++ b/.github/skills/internal-gateway-critical-master/fixtures/critical_output_valid_premortem.md @@ -0,0 +1,5 @@ +🎯 **Piano:** Spostare tutti i controlli dalla CI ai computer degli sviluppatori. +⚠️ **Critica:** Perderemmo la prova centrale perché i controlli locali non producono un registro condiviso. +💥 **Rischio:** Alcuni repository potrebbero saltare i controlli senza che nessuno se ne accorga. +✅ **Consiglio:** Mantenere la CI finché non esiste un controllo centrale equivalente. +❓ **Da chiarire:** Cosa sostituirà ufficialmente i log della CI? diff --git a/.github/skills/internal-gateway-critical-master/fixtures/routing_cases.json b/.github/skills/internal-gateway-critical-master/fixtures/routing_cases.json new file mode 100644 index 00000000..cbd3e510 --- /dev/null +++ b/.github/skills/internal-gateway-critical-master/fixtures/routing_cases.json @@ -0,0 +1,22 @@ +[ + { + "id": "pressure-test-plan", + "prompt": "Pressure-test this migration plan and expose the assumption most likely to invalidate it.", + "expected_owner": "internal-gateway-critical-master" + }, + { + "id": "defect-first-skill-review", + "prompt": "Review this concrete skill bundle for defects, contract drift, and missing validation.", + "expected_owner": "internal-review-code" + }, + { + "id": "shape-unclear-idea", + "prompt": "Help me compare options and shape an unclear repository automation idea.", + "expected_owner": "internal-gateway-idea" + }, + { + "id": "execute-local-edit", + "prompt": "Apply this already-defined local metadata edit and run its focused validator.", + "expected_owner": "internal-gateway-simple-task" + } +] diff --git a/.github/skills/internal-gateway-critical-master/references/output-contract.md b/.github/skills/internal-gateway-critical-master/references/output-contract.md index 313ad135..4fa1267b 100644 --- a/.github/skills/internal-gateway-critical-master/references/output-contract.md +++ b/.github/skills/internal-gateway-critical-master/references/output-contract.md @@ -1,76 +1,54 @@ # Output Contract -Use this reference to produce a compact, consistent critical challenge deliverable. +Use this reference to produce the public projection of a critical challenge. -## Required fields - -| Field | Description | Max length | -| --- | --- | --- | -| `summary` | One paragraph: what is being challenged and why it matters now. | 75 words | -| `findings` | 1-3 findings. Each finding uses the sub-fields below. | 3 items | -| `finding.objection` | Strongest objection or assumption gap. | 30 words | -| `finding.impact` | Why it matters now. | 30 words | -| `finding.evidence` | Repository evidence, inference, or named uncertainty. | 30 words | -| `finding.mitigation` | Condition or action required before execution resumes. | 30 words | -| `finding.reframe` | Optional lateral reframe. | 25 words | -| `finding.question` | Optional single root question when the answer would materially change the critique. | 25 words | -| `synthesis` | Result of the Final Consistency Gate, including defense classification, strongest defense, and remaining vulnerability when user defenses exist. | 100 words | -| `outcome` | Exactly one value from `## Outcome meanings` in `SKILL.md`. | 1 value | - -## Budget - -- Total output target: **600 words or fewer**. -- If the material demands more, split the work into another critical cycle. -- Do not pad findings to reach 3; one strong finding is better than three weak ones. - -## Output template +## Adaptive card layout ```markdown -## Summary - - - -## Findings - -### 1. +🎯 **:** +⚠️ **:** +💥 **:** +✅ **:** +❓ **:** +``` -- **Impact:** -- **Evidence:** -- **Mitigation:** -- **Reframe:** -- **Question:** +## Rules -## Synthesis +- Three required lines: 🎯, ⚠️, ✅. +- Two optional lines: 💥 and ❓. +- Exact order: 🎯, ⚠️, [💥], ✅, [❓]. +- Adaptive length: three to five content lines. +- 💥 only when there is a material risk. +- ❓ only when the answer could change the advice. +- No visible canonical outcome codes or technical classification labels. +- No headings, preamble, appendix, or old report sections. +- Visible labels match the user's language. Emoji identify fields for validation. +- The critique line (⚠️) states both what is wrong and one concrete reason. - +## Examples -## Outcome +### Minimal -`` +```markdown +🎯 **Plan:** Move validation from CI to developer machines. +⚠️ **Critique:** Central proof disappears because local checks do not create a shared record. +✅ **Advice:** Keep CI until an equivalent central control exists. ``` -## Example +### Complex (Italian) ```markdown -## Summary - -We are challenging a proposal to move validation logic from CI into a pre-commit hook. The change matters now because it affects every contributor's workflow and could hide failures from the central audit log. - -## Findings - -### 1. The audit trail weakens - -- **Impact:** Central CI logs become incomplete for compliance reviews. -- **Evidence:** `inference` — no replacement logging is described. -- **Mitigation:** Add a signed attestation step before the hook is enabled. -- **Reframe:** Treat local validation as an early filter, not a replacement for CI. (Analogy: like a spell-checker that does not replace a full code review.) -- **Question:** Which central audit record replaces the CI validation log? - -## Synthesis - -The strongest risk is compliance visibility, not implementation effort. The proposal can work if the mitigation is accepted. +🎯 **Piano:** Spostare tutti i controlli dalla CI ai computer degli sviluppatori. +⚠️ **Critica:** Perderemmo la prova centrale perché i controlli locali non producono un registro condiviso. +💥 **Rischio:** Alcuni repository potrebbero saltare i controlli senza che nessuno se ne accorga. +✅ **Consiglio:** Mantenere la CI finché non esiste un controllo centrale equivalente. +❓ **Da chiarire:** Cosa sostituirà ufficialmente i log della CI? +``` -## Outcome +## Validator limits -`accept-with-risk` -``` +- Per-line word budget: 30 words. +- Total word budget: 100 words (overridable via `--max-words`). +- Legacy H2 sections are rejected. +- Non-empty prose outside the card is rejected. +- Semantic reason quality (whether the critique actually states a useful reason) is prompt-governed, not validator-enforced. diff --git a/.github/skills/internal-gateway-critical-master/scripts/critical_master.py b/.github/skills/internal-gateway-critical-master/scripts/critical_master.py index f5f70e4f..23ca9015 100644 --- a/.github/skills/internal-gateway-critical-master/scripts/critical_master.py +++ b/.github/skills/internal-gateway-critical-master/scripts/critical_master.py @@ -15,9 +15,9 @@ { "reformulate-plan", "de-escalate-to-simple", - "execute-clear-next-step", + "route-to-execution-owner", "review-evidence", - "continue-critical", + "continue-critical-with-new-evidence", "accept-with-risk", } ) @@ -26,30 +26,25 @@ {"confirmed", "inference", "estimate"} ) -REQUIRED_SECTIONS: tuple[str, ...] = ( - "Summary", - "Findings", - "Synthesis", - "Outcome", +ALLOWED_EVIDENCE_QUALITY: frozenset[str] = frozenset( + {"strong", "partial", "weak"} ) -REQUIRED_FINDING_FIELDS: tuple[str, ...] = ( - "Impact", - "Evidence", - "Mitigation", -) +ALLOWED_LIKELIHOODS: frozenset[str] = frozenset({"high", "medium", "low"}) -OPTIONAL_FINDING_FIELDS: tuple[str, ...] = ("Reframe", "Question") +ALLOWED_DEFENSE_VALUES: frozenset[str] = frozenset( + {"none", "resolves", "narrows", "accepts-risk", "unanswered"} +) -SUMMARY_MAX_WORDS = 75 -SYNTHESIS_MAX_WORDS = 100 -FINDING_OBJECTION_MAX_WORDS = 30 -FINDING_FIELD_MAX_WORDS = 30 -FINDING_REFRAME_MAX_WORDS = 25 -FINDING_QUESTION_MAX_WORDS = 25 TOTAL_MAX_WORDS = 600 -MIN_FINDINGS = 1 -MAX_FINDINGS = 3 + +REQUIRED_CARD_MARKERS = ("🎯", "⚠️", "✅") +OPTIONAL_CARD_MARKERS = ("💥", "❓") +CARD_MARKER_ORDER = ("🎯", "⚠️", "💥", "✅", "❓") +CARD_MIN_LINES = 3 +CARD_MAX_LINES = 5 +CARD_TOTAL_MAX_WORDS = 100 +CARD_LINE_MAX_WORDS = 30 @dataclass(frozen=True) @@ -88,109 +83,42 @@ def _strip_code_fences(text: str) -> str: return re.sub(r"```.*?```", "", text, flags=re.DOTALL) -_HEADING_PATTERN = re.compile(r"^(#{1,6})\s+(.*?)\s*$", re.MULTILINE) - +_CARD_LINE_PATTERN = re.compile( + r"^(🎯|⚠️|💥|✅|❓)\s+\*\*([^*]+):\*\*\s+(.+?)\s*$" +) -def parse_markdown_sections(text: str) -> dict[str, str]: - """Parse a Markdown document into ``{section_title: body}``.""" - matches = list(_HEADING_PATTERN.finditer(text)) - sections: dict[str, str] = {} - if not matches: - return sections - top_level_indices = [ - index for index, match in enumerate(matches) if len(match.group(1)) == 2 - ] - for list_index, match_index in enumerate(top_level_indices): - match = matches[match_index] - title = match.group(2).strip() - if title in sections: - continue - start = match.end() - if list_index + 1 < len(top_level_indices): - end = matches[top_level_indices[list_index + 1]].start() - else: - end = len(text) - body = text[start:end] - sections[title] = body.rstrip() - return sections +@dataclass(frozen=True) +class CardLine: + marker: str + label: str + content: str + raw: str @dataclass(frozen=True) -class ParsedFinding: - """One finding extracted from a rendered Markdown output.""" - - heading: str - body: str - has_impact: bool - has_evidence: bool - has_mitigation: bool - has_reframe: bool - has_question: bool - evidence_class: str | None - raw_lines: tuple[str, ...] - - -_FINDING_HEADING_PATTERN = re.compile(r"^(#{3,6})\s+(\d+)\.\s+(.*?)\s*$", re.MULTILINE) - - -def parse_findings(findings_body: str) -> list[ParsedFinding]: - """Parse the ``## Findings`` section body into individual finding records.""" - matches = list(_FINDING_HEADING_PATTERN.finditer(findings_body)) - parsed: list[ParsedFinding] = [] - for index, match in enumerate(matches): - heading = f"{match.group(2)}. {match.group(3)}" - start = match.end() - end = matches[index + 1].start() if index + 1 < len(matches) else len(findings_body) - body = findings_body[start:end].rstrip() - lower_body = body.lower() - has_impact = "**impact:**" in lower_body - has_evidence = "**evidence:**" in lower_body - has_mitigation = "**mitigation:**" in lower_body - has_reframe = "**reframe:**" in lower_body - has_question = "**question:**" in lower_body - evidence_class = _extract_evidence_class(body) - parsed.append( - ParsedFinding( - heading=heading, - body=body, - has_impact=has_impact, - has_evidence=has_evidence, - has_mitigation=has_mitigation, - has_reframe=has_reframe, - has_question=has_question, - evidence_class=evidence_class, - raw_lines=tuple(body.splitlines()), +class CriticalCard: + lines: tuple[CardLine, ...] + + @property + def by_marker(self) -> dict[str, CardLine]: + return {line.marker: line for line in self.lines} + + +def parse_critical_card(text: str) -> CriticalCard: + lines: list[CardLine] = [] + for raw_line in text.splitlines(): + match = _CARD_LINE_PATTERN.match(raw_line.strip()) + if match: + lines.append( + CardLine( + marker=match.group(1), + label=match.group(2).strip(), + content=match.group(3).strip(), + raw=raw_line, + ) ) - ) - return parsed - - -_CLAIM_CLASS_PATTERN = re.compile( - r"\*\*[Ee]vidence:\*\*\s*`?(confirmed|inference|estimate)`?", - re.IGNORECASE, -) - - -def _extract_evidence_class(body: str) -> str | None: - match = _CLAIM_CLASS_PATTERN.search(body) - if not match: - return None - return match.group(1).lower() - - -_OUTCOME_PATTERN = re.compile(r"`([^`]+)`") - - -def extract_outcome_value(outcome_text: str) -> str | None: - """Extract the single backtick-wrapped outcome from the Outcome section.""" - cleaned = outcome_text.strip() - if not cleaned: - return None - match = _OUTCOME_PATTERN.search(cleaned) - if not match: - return None - return match.group(1).strip() + return CriticalCard(lines=tuple(lines)) def validate_outcome_value(value: str) -> bool: @@ -198,32 +126,24 @@ def validate_outcome_value(value: str) -> bool: return value in ALLOWED_OUTCOMES -def classify_claim_class(body: str) -> str | None: - """Return the claim class extracted from an Evidence bullet, or None.""" - return _extract_evidence_class(body) - - __all__ = [ "ALLOWED_CLAIM_CLASSES", + "ALLOWED_DEFENSE_VALUES", + "ALLOWED_EVIDENCE_QUALITY", + "ALLOWED_LIKELIHOODS", "ALLOWED_OUTCOMES", + "CARD_LINE_MAX_WORDS", + "CARD_MARKER_ORDER", + "CARD_MAX_LINES", + "CARD_MIN_LINES", + "CARD_TOTAL_MAX_WORDS", + "CardLine", + "CriticalCard", "Finding", - "FINDING_FIELD_MAX_WORDS", - "FINDING_OBJECTION_MAX_WORDS", - "FINDING_QUESTION_MAX_WORDS", - "FINDING_REFRAME_MAX_WORDS", - "MAX_FINDINGS", - "MIN_FINDINGS", - "OPTIONAL_FINDING_FIELDS", - "ParsedFinding", - "REQUIRED_FINDING_FIELDS", - "REQUIRED_SECTIONS", - "SUMMARY_MAX_WORDS", - "SYNTHESIS_MAX_WORDS", + "OPTIONAL_CARD_MARKERS", + "REQUIRED_CARD_MARKERS", "TOTAL_MAX_WORDS", - "classify_claim_class", "count_words", - "extract_outcome_value", - "parse_findings", - "parse_markdown_sections", + "parse_critical_card", "validate_outcome_value", ] diff --git a/.github/skills/internal-gateway-critical-master/scripts/validate_critical_output.py b/.github/skills/internal-gateway-critical-master/scripts/validate_critical_output.py index d8760409..83c9ccce 100644 --- a/.github/skills/internal-gateway-critical-master/scripts/validate_critical_output.py +++ b/.github/skills/internal-gateway-critical-master/scripts/validate_critical_output.py @@ -3,8 +3,8 @@ Usage examples: python3 scripts/validate_critical_output.py --file fixtures/critical_output_valid.md - python3 scripts/validate_critical_output.py --file fixtures/critical_output_invalid_missing_section.md --format json - python3 scripts/validate_critical_output.py --file fixtures/critical_output_advisory.md --strict + python3 scripts/validate_critical_output.py --file fixtures/critical_output_valid.md --strict + python3 scripts/validate_critical_output.py --file fixtures/critical_output_valid.md --format json """ from __future__ import annotations @@ -17,25 +17,14 @@ from typing import Protocol, TypeVar from critical_master import ( - ALLOWED_CLAIM_CLASSES, - ALLOWED_OUTCOMES, - FINDING_FIELD_MAX_WORDS, - FINDING_OBJECTION_MAX_WORDS, - FINDING_QUESTION_MAX_WORDS, - FINDING_REFRAME_MAX_WORDS, - MAX_FINDINGS, - MIN_FINDINGS, - REQUIRED_FINDING_FIELDS, - REQUIRED_SECTIONS, - SUMMARY_MAX_WORDS, - SYNTHESIS_MAX_WORDS, - TOTAL_MAX_WORDS, + CARD_LINE_MAX_WORDS, + CARD_MARKER_ORDER, + CARD_MAX_LINES, + CARD_MIN_LINES, + CARD_TOTAL_MAX_WORDS, Finding, count_words, - extract_outcome_value, - parse_findings, - parse_markdown_sections, - validate_outcome_value, + parse_critical_card, ) @@ -47,6 +36,8 @@ def to_dict(self) -> dict[str, object]: ... FindingT = TypeVar("FindingT", bound=FindingLike) +_H2_PATTERN = re.compile(r"^##\s+", re.MULTILINE) + def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( @@ -65,7 +56,7 @@ def parse_args() -> argparse.Namespace: parser.add_argument( "--max-words", type=int, - default=TOTAL_MAX_WORDS, + default=CARD_TOTAL_MAX_WORDS, help="Override the total output word limit.", ) parser.add_argument( @@ -114,18 +105,15 @@ def render_json(data: object) -> str: return json.dumps(data, indent=2, sort_keys=True) -def log_info(message: str) -> None: - print(f"INFO: {message}", flush=True) - - -def log_warn(message: str) -> None: - print(f"WARN: {message}", flush=True) - - def main() -> int: args = parse_args() if args.file: - text = Path(args.file).read_text(encoding="utf-8") + path = Path(args.file) + try: + text = path.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError) as exc: + print(f"ERROR: cannot read '{path}': {exc}", file=sys.stderr) + return 2 else: text = sys.stdin.read() @@ -138,250 +126,167 @@ def main() -> int: return 1 if should_fail(findings, strict=args.strict) else 0 -def validate_output(text: str, *, max_words: int = TOTAL_MAX_WORDS) -> list[Finding]: - """Run every output-contract check and return a list of Findings.""" +def validate_output( + text: str, *, max_words: int = CARD_TOTAL_MAX_WORDS +) -> list[Finding]: findings: list[Finding] = [] - sections = parse_markdown_sections(text) - - for required in REQUIRED_SECTIONS: - if required not in sections: - findings.append( - Finding( - severity="blocking", - code=f"missing-section-{_section_slug(required)}", - path="(output)", - message=f"Required section '## {required}' is missing.", - suggestion=( - f"Add a '## {required}' section. " - "See references/output-contract.md for the template." - ), - ) - ) - - summary_body = sections.get("Summary", "") - synthesis_body = sections.get("Synthesis", "") - findings_body = sections.get("Findings", "") - outcome_body = sections.get("Outcome", "") - - summary_words = count_words(summary_body) - if summary_body and summary_words > SUMMARY_MAX_WORDS: - findings.append( - Finding( - severity="non-blocking", - code="summary-word-limit", - path="## Summary", - message=( - f"Summary has {summary_words} words; limit is {SUMMARY_MAX_WORDS}." - ), - suggestion="Compress the summary paragraph.", - extras={"words": summary_words, "limit": SUMMARY_MAX_WORDS}, - ) - ) - - synthesis_words = count_words(synthesis_body) - if synthesis_body and synthesis_words > SYNTHESIS_MAX_WORDS: - findings.append( - Finding( - severity="non-blocking", - code="synthesis-word-limit", - path="## Synthesis", - message=( - f"Synthesis has {synthesis_words} words; limit is {SYNTHESIS_MAX_WORDS}." - ), - suggestion="Compress the synthesis paragraph.", - extras={"words": synthesis_words, "limit": SYNTHESIS_MAX_WORDS}, - ) - ) - parsed_findings = parse_findings(findings_body) - finding_count = len(parsed_findings) - if finding_count < MIN_FINDINGS or finding_count > MAX_FINDINGS: + if _H2_PATTERN.search(text): findings.append( Finding( severity="blocking", - code="finding-count-out-of-range", - path="## Findings", - message=( - f"Found {finding_count} findings; expected {MIN_FINDINGS}-{MAX_FINDINGS}." - ), - suggestion=( - "Adjust the count to fit the contract: one strong finding is " - "better than three weak ones." - ), - extras={"count": finding_count, "min": MIN_FINDINGS, "max": MAX_FINDINGS}, - ) - ) - - for parsed in parsed_findings: - _check_finding(parsed, findings) - - outcome_value = extract_outcome_value(outcome_body) - if outcome_value is None: - findings.append( - Finding( - severity="blocking", - code="missing-outcome-value", - path="## Outcome", - message="Outcome section is missing a backtick-wrapped value.", - suggestion="Wrap the outcome value in single backticks, e.g. `accept-with-risk`.", + code="legacy-section-format", + path="(output)", + message="Legacy H2 sections are not allowed in the new card format.", + suggestion="Use the emoji card layout from references/output-contract.md.", ) ) - elif not validate_outcome_value(outcome_value): + return findings + + lines = text.splitlines() + card = parse_critical_card(text) + non_empty_prose = [ + line for line in lines if line.strip() and line.strip() not in { + cl.raw.strip() for cl in card.lines + } + ] + if non_empty_prose: findings.append( Finding( severity="blocking", - code="invalid-outcome-value", - path="## Outcome", - message=( - f"Outcome value '{outcome_value}' is not in the allowed set." - ), - suggestion=( - f"Use one of: {', '.join(sorted(ALLOWED_OUTCOMES))}." - ), - extras={"allowed": sorted(ALLOWED_OUTCOMES)}, + code="unexpected-content-line", + path="(output)", + message="Non-empty line does not match any card marker.", + suggestion="Each content line must start with an emoji marker.", ) ) - total_words = count_words(text) - if total_words > max_words: - findings.append( - Finding( - severity="non-blocking", - code="total-word-limit", - path="(output)", - message=( - f"Total output is {total_words} words; limit is {max_words}." - ), - suggestion="Compress the deliverable or split into multiple cycles.", - extras={"words": total_words, "limit": max_words}, + markers_present = [cl.marker for cl in card.lines] + seen_markers: set[str] = set() + for marker in markers_present: + if marker in seen_markers: + findings.append( + Finding( + severity="blocking", + code="duplicate-marker", + path="(output)", + message=f"Marker '{marker}' appears more than once.", + suggestion="Each marker is allowed at most once.", + ) ) - ) + seen_markers.add(marker) - return findings + for required in ("🎯", "⚠️", "✅"): + if required not in seen_markers: + code_map = {"🎯": "missing-plan", "⚠️": "missing-critique", "✅": "missing-advice"} + findings.append( + Finding( + severity="blocking", + code=code_map[required], + path="(output)", + message=f"Required marker '{required}' is missing.", + suggestion="Add the missing emoji line to the card.", + ) + ) + if "💥" in seen_markers and "✅" in seen_markers: + risk_idx = markers_present.index("💥") + advice_idx = markers_present.index("✅") + if risk_idx > advice_idx: + findings.append( + Finding( + severity="blocking", + code="card-line-order", + path="(output)", + message="Risk marker 💥 must appear before advice ✅.", + suggestion="Move 💥 before ✅.", + ) + ) -def _check_finding(parsed, findings: list[Finding]) -> None: - path = f"## Findings :: {parsed.heading}" - for required in REQUIRED_FINDING_FIELDS: - if not getattr(parsed, f"has_{required.lower()}"): + if "❓" in seen_markers and "✅" in seen_markers: + question_idx = markers_present.index("❓") + advice_idx = markers_present.index("✅") + if question_idx < advice_idx: findings.append( Finding( severity="blocking", - code=f"missing-finding-field-{required.lower()}", - path=path, - message=f"Finding is missing **'{required}:'** field.", - suggestion=( - f"Add a '**{required}:**' bullet under the finding heading." - ), + code="card-line-order", + path="(output)", + message="Question marker ❓ must appear after advice ✅.", + suggestion="Move ❓ after ✅.", ) ) - if parsed.evidence_class is None: - findings.append( - Finding( - severity="blocking", - code="missing-claim-class", - path=path, - message=( - "Evidence field must declare a claim class " - f"({', '.join(sorted(ALLOWED_CLAIM_CLASSES))})." - ), - suggestion=( - "Append the class to the Evidence bullet, " - "e.g. '**Evidence:** `inference` — ...'." - ), + + expected_order = [m for m in CARD_MARKER_ORDER if m in seen_markers] + actual_order = [] + for marker in markers_present: + if marker not in actual_order: + actual_order.append(marker) + if expected_order != actual_order: + already_reported_order = any(f.code == "card-line-order" for f in findings) + if not already_reported_order: + findings.append( + Finding( + severity="blocking", + code="card-line-order", + path="(output)", + message="Card markers are not in canonical order.", + suggestion="Use order: 🎯, ⚠️, [💥], ✅, [❓].", + ) ) - ) - elif parsed.evidence_class not in ALLOWED_CLAIM_CLASSES: + + line_count = len(card.lines) + if line_count < CARD_MIN_LINES or line_count > CARD_MAX_LINES: findings.append( Finding( severity="blocking", - code="invalid-claim-class", - path=path, - message=( - f"Claim class '{parsed.evidence_class}' is not in the allowed set." - ), - suggestion=( - f"Use one of: {', '.join(sorted(ALLOWED_CLAIM_CLASSES))}." - ), - ) - ) - objection_text = re.sub(r"^\d+\.\s+", "", parsed.heading) - objection_words = count_words(objection_text) - if objection_words > FINDING_OBJECTION_MAX_WORDS: - findings.append( - Finding( - severity="non-blocking", - code="finding-objection-word-limit", - path=path, - message=( - f"Objection has {objection_words} words; " - f"limit is {FINDING_OBJECTION_MAX_WORDS}." - ), - suggestion="Shorten the objection heading.", - extras={"words": objection_words, "limit": FINDING_OBJECTION_MAX_WORDS}, + code="card-line-count", + path="(output)", + message=f"Card has {line_count} lines; expected {CARD_MIN_LINES}-{CARD_MAX_LINES}.", + suggestion="Use three to five content lines.", + extras={"count": line_count, "min": CARD_MIN_LINES, "max": CARD_MAX_LINES}, ) ) - for field_name in ("Impact", "Evidence", "Mitigation"): - words = _field_word_count(parsed.body, field_name) - if words > FINDING_FIELD_MAX_WORDS: - findings.append( - Finding( - severity="non-blocking", - code=f"finding-field-word-limit-{field_name.lower()}", - path=path, - message=( - f"'{field_name}' has {words} words; " - f"limit is {FINDING_FIELD_MAX_WORDS}." - ), - suggestion=f"Shorten the '{field_name}' bullet.", - extras={ - "words": words, - "limit": FINDING_FIELD_MAX_WORDS, - }, - ) - ) - if parsed.has_reframe: - words = _field_word_count(parsed.body, "Reframe") - if words > FINDING_REFRAME_MAX_WORDS: + + for cl in card.lines: + if not cl.content.strip(): findings.append( Finding( - severity="non-blocking", - code="finding-reframe-word-limit", - path=path, - message=( - f"Reframe has {words} words; limit is {FINDING_REFRAME_MAX_WORDS}." - ), - suggestion="Shorten the optional reframe.", - extras={"words": words, "limit": FINDING_REFRAME_MAX_WORDS}, + severity="blocking", + code="unexpected-content-line", + path=f"(marker {cl.marker})", + message="Card line has empty content after label.", + suggestion="Add content after the bold label.", ) ) - if parsed.has_question: - words = _field_word_count(parsed.body, "Question") - if words > FINDING_QUESTION_MAX_WORDS: + continue + line_words = count_words(cl.content) + if line_words > CARD_LINE_MAX_WORDS: findings.append( Finding( severity="non-blocking", - code="finding-question-word-limit", - path=path, - message=( - f"Question has {words} words; limit is {FINDING_QUESTION_MAX_WORDS}." - ), - suggestion="Shorten the optional root question.", - extras={"words": words, "limit": FINDING_QUESTION_MAX_WORDS}, + code="card-line-word-limit", + path=f"(marker {cl.marker})", + message=f"Line has {line_words} words; limit is {CARD_LINE_MAX_WORDS}.", + suggestion="Shorten the line.", + extras={"words": line_words, "limit": CARD_LINE_MAX_WORDS}, ) ) + total_words = count_words(text) + if total_words > max_words: + findings.append( + Finding( + severity="non-blocking", + code="total-word-limit", + path="(output)", + message=f"Total output is {total_words} words; limit is {max_words}.", + suggestion="Compress the card.", + extras={"words": total_words, "limit": max_words}, + ) + ) -def _field_word_count(body: str, field_name: str) -> int: - pattern = re.compile(rf"\*\*{field_name}:\*\*\s*(.*?)(?=\n\s*-\s*\*\*|\Z)", re.DOTALL) - match = pattern.search(body) - if not match: - return 0 - return count_words(match.group(1)) - - -def _section_slug(title: str) -> str: - return title.lower().replace(" ", "-") + return findings def _compact_payload(findings: list[Finding]) -> dict[str, object]: @@ -405,7 +310,7 @@ def render_text(findings: list[Finding]) -> None: return for finding in findings: marker = "BLOCKING" if finding.severity == "blocking" else "advisory" - log_warn(f"[{marker}] {finding.path} :: {finding.code} :: {finding.message}") + print(f"[{marker}] {finding.path} :: {finding.code} :: {finding.message}") print(f" Suggestion: {finding.suggestion}") diff --git a/.github/skills/internal-gateway-execute-plans/SKILL.md b/.github/skills/internal-gateway-execute-plans/SKILL.md index 3c61bc82..ab2bc69b 100644 --- a/.github/skills/internal-gateway-execute-plans/SKILL.md +++ b/.github/skills/internal-gateway-execute-plans/SKILL.md @@ -1,74 +1,75 @@ --- name: internal-gateway-execute-plans -description: Use when executing or resuming an approved repository-owned retained plan under tmp/superpowers/. +description: "Use when executing or resuming an approved repository-owned retained plan under tmp/superpowers/plans/." --- # Internal Gateway Execute Plans -## Referenced skills +## Bundle References + +- `references/execution-contract.md` — repository hooks around the delegated execution loop. +- `references/status-contract.md` — status transition table, required headings, and exact sibling filenames. +- `scripts/plan_execution.py` — read-only stdlib-only CLI for plan binding, status shape, resume safety, and completion readiness. -- `superpowers-executing-plans`: required owner for task-by-task plan execution. -- `internal-tdd`: repository-owned test-first route for tasks that change executable or evaluable behavior. -- `superpowers-verification-before-completion`: on-demand evidence gate before completion claims. -- `addyosmani-code-simplification`: plan-bound method owner only when the current approved plan task explicitly requires behavior-preserving simplification or records an approved review remediation. +## Referenced skills -Thin repository wrapper for approved retained-plan execution. It owns only -repo-local start, task-transition, stop, and status-file policy; -`superpowers-executing-plans` owns the execution loop. +- `/superpowers-executing-plans` owns critical plan review, todo tracking, task execution, and its core stop behavior. +- `/internal-tdd` owns executable-behavior test-first guidance at the local task gate. +- `/superpowers-verification-before-completion` owns final evidence before completion claims. +- `/addyosmani-code-simplification` is conditional and may be loaded only when the approved task explicitly authorizes simplification. ## When to use -- Execute or resume an approved retained plan under `tmp/superpowers/`. -- Apply repo-local closeout through `..md`. +- Execute or resume an approved retained plan under `tmp/superpowers/plans/`. +- Apply repository-local preflight, task hooks, status handling, and closeout around the delegated core loop. ## When not to use - Writing, reformulating, reviewing, or challenging a plan. - Running same-chat work that is not driven by an approved retained plan. -- Changing `superpowers-executing-plans`; imported Superpowers behavior stays read-only unless the user explicitly changes scope. +- Changing imported execution behavior or replacing the delegated core workflow. -## Execution Discipline +## Gateway boundary -- Plan-bound: follow the approved retained plan and stop when it is no longer executable as written. -- DRY: do not duplicate execution workflow, validation reporting, status policy, or source logic owned elsewhere. -- YAGNI: do not add speculative helpers, abstractions, configuration, or future proofing beyond the approved plan. -- KISS: choose the simplest coherent change that satisfies the plan and stays readable. -- Separation of Concerns: keep planning, execution, validation evidence, closeout status, review, critique, and domain implementation separate. -- Single responsibility: each new helper, section, or code path must have one active-task reason. -- Tight feedback: treat the smallest coherent change set as one execution unit, then run its focused check. Coupled edits such as signature, callers, and body belong to one unit; unrelated plan tasks do not. -- Test first: when a task changes executable or evaluable behavior, load `internal-tdd` before its first implementation edit and follow the selected posture. -- Fail fast on drift: stop and record the defect when the plan forces confusing code, duplicated logic, missing validation, owner conflict, or scope drift. -- Scoped simplification: load `addyosmani-code-simplification` only when the current approved plan task explicitly requires behavior-preserving code simplification or records an approved review remediation; never introduce it as cleanup outside the approved plan. +The gateway is a repository-owned extension of `/superpowers-executing-plans`. The +delegated owner supplies the core review, todo, execution, and stop loop. Keep +only these local responsibilities here: -## Contract +- bind the exact plan path and explicit approval state; +- compute the SHA-256 fingerprint and run dirty-worktree preflight; +- apply task-level `/internal-tdd` and evidence hooks; +- enforce the no-Git-mutation policy; +- replace the exact `DONE`, `PARTIAL`, `BLOCKED`, or `NEEDS_REVIEW` sibling; +- run resume and completion checks through `scripts/plan_execution.py`. -1. Confirm the retained plan folder, plan basename, and approval to execute. -2. Read only target, anti-scope, validation path, stop conditions, and first executable task. -3. On resume, verify any existing `..md`; do not resume from `DONE` unless fresh evidence invalidates it. -4. Announce this gateway, then load `superpowers-executing-plans` for critical plan review, todos, task execution, and its stop rules. -5. Before each task, name its observable outcome, dependency set, and focused validation. For interface or signature changes, identify affected callers, implementations, tests, and contracts before editing. -6. When the approved task authorizes simplification, establish the passing behavior baseline, load `addyosmani-code-simplification`, keep the refactor inside the task scope, and rerun the same focused validation afterward. -7. Execute one smallest coherent change set. Complete all known coupled edits, perform a consistency pass across the dependency set, then run the focused validation before widening the task. -8. Before marking a task complete or starting the next task, require fresh passing task-level evidence. If the check is missing, not run, or failing, keep the task in progress or blocked; do not claim completion. -9. Preserve compact state with targeted rereads. Summarize large validator output by command, exit code, material counts, changed files, and exact gaps. -10. Stop on scope drift, destructive action, owner conflict, missing validation path, human approval need, secret exposure risk, or repeated non-improving failures. -11. Before final response or pause, replace any older sibling `.*.md` status file for the same plan basename, then write exactly one sibling status file named `..md`. +## Delegation checkpoints -## Status closeout +Before loading `/superpowers-executing-plans`, bind the retained plan, record +approval, fingerprint the plan, and capture the workspace baseline. At each +task boundary, load `/internal-tdd` when the task changes executable or +evaluable behavior and require its red-first evidence before implementation. +After each delegated task, run the plan's focused validation and retain fresh +evidence. Load `/superpowers-verification-before-completion` before any positive +completion claim; load `/addyosmani-code-simplification` only when explicitly +authorized by the plan. -Supported statuses are `DONE`, `BLOCKED`, `PARTIAL`, and `NEEDS_REVIEW`. -Required headings are `## Status`, `## Reason`, `## Completed`, -`## Remaining`, `## Validation`, `## Next`, and `## Resume Notes`. +On pause or resume, preserve the plan fingerprint and use the status and resume +checks from `scripts/plan_execution.py`. At closeout, run the required broader +validation, verify `git diff --check`, and write exactly one status sibling +according to `references/status-contract.md`. -Before claiming `DONE`, load `superpowers-verification-before-completion` and present fresh passing evidence. +## No-Commit Rule -Use `DONE` only when every task passed its transition gate, all in-scope work is complete, and required broader validation has fresh passing evidence. A final broad check does not retroactively validate skipped task gates. For any gap, use the status that best explains the remaining action and record the exact evidence needed to resume or finish. - -Do not create `done-*`, `completion-report.md`, `evidence-envelope.md`, or `-plan-state.md` as new closeout artifacts. +Do not run `git add`, `git commit`, `git push`, `git merge`, or another Git +mutation while executing, pausing, or closing out a plan. Leave executed +changes uncommitted for the user to review. If a retained plan contains Git +mutation steps, skip them and record the plan drift in the status sibling. ## Validation -- Pressure-check that a task cannot become complete without fresh focused evidence and that an interface change requires dependency consistency. -- Confirm no live repository references point to removed bundle files. - `git diff --check` -- Confirm simplification loads only from an explicitly authorizing plan task, preserves task scope, and reruns the same focused validation used for the baseline. +- `python3 scripts/plan_execution.py preflight --format compact` +- `python3 scripts/plan_execution.py status-check --format compact` +- `python3 scripts/plan_execution.py resume-check --format compact` +- `python3 scripts/plan_execution.py completion-check --format compact` +- Confirm no live repository references point to removed bundle files. diff --git a/.github/skills/internal-gateway-execute-plans/agents/openai.yaml b/.github/skills/internal-gateway-execute-plans/agents/openai.yaml index 20c5f4e9..ed1198e3 100644 --- a/.github/skills/internal-gateway-execute-plans/agents/openai.yaml +++ b/.github/skills/internal-gateway-execute-plans/agents/openai.yaml @@ -1,4 +1,4 @@ interface: display_name: "Internal Gateway Execute Plans" - short_description: "Gateway for retained-plan execution" - default_prompt: "Use $internal-gateway-execute-plans for approved retained-plan execution or resume under tmp/superpowers/. Wrap, then delegate the execution loop to superpowers-executing-plans. Keep superpowers-executing-plans read-only. Before final response or pause, replace any older sibling .*.md status file for the same plan basename, then write exactly one sibling status file named ..md using DONE, BLOCKED, PARTIAL, or NEEDS_REVIEW." + short_description: "Repository gateway for approved retained-plan execution" + default_prompt: "Use /internal-gateway-execute-plans for approved retained-plan execution or resume under tmp/superpowers/plans/. Load /superpowers-executing-plans for critical plan review, todos, task execution, and core stop behavior. At the local gates load /internal-tdd for executable-behavior test-first work, /superpowers-verification-before-completion before final evidence, and /addyosmani-code-simplification only when the approved task authorizes simplification. Bind the exact plan path, compute the plan fingerprint with SHA-256, check workspace overlap, preserve local routing and status rules, require fresh task-level evidence, write exactly one allowed status sibling before pause or closeout, and enforce no Git mutation throughout." diff --git a/.github/skills/internal-gateway-execute-plans/fixtures/invalid-status.md b/.github/skills/internal-gateway-execute-plans/fixtures/invalid-status.md new file mode 100644 index 00000000..29bd790f --- /dev/null +++ b/.github/skills/internal-gateway-execute-plans/fixtures/invalid-status.md @@ -0,0 +1,11 @@ +## Status + +`UNKNOWN_STATE` + +## Plan + +`valid-plan.md` + +## Reason + +Invalid fixture. diff --git a/.github/skills/internal-gateway-execute-plans/fixtures/valid-plan.PARTIAL.md b/.github/skills/internal-gateway-execute-plans/fixtures/valid-plan.PARTIAL.md new file mode 100644 index 00000000..f5d2aa0e --- /dev/null +++ b/.github/skills/internal-gateway-execute-plans/fixtures/valid-plan.PARTIAL.md @@ -0,0 +1,44 @@ +## Status + +`PARTIAL` + +## Plan + +`valid-plan.md` + +## Plan Fingerprint + +`sha256:137de56698597d2a33f3644f6ebc693139c571f485ce4a7460d2dd4ff7d6c3ca` + +## Reason + +Task 1 is complete; remaining tasks pending execution. + +## Workspace Baseline + +- Branch: main +- Dirty files: none in scope + +## Files Changed + +- `fixtures/valid-plan.PARTIAL.md` + +## Completed + +- Task 1: Validate CLI — pytest passed + +## Remaining + +- Task 2: Integration test + +## Validation + +- `pytest -q tests/fixture/` — passed + +## Next + +Execute Task 2 + +## Resume Notes + +No blockers; continue from Task 2. diff --git a/.github/skills/internal-gateway-execute-plans/fixtures/valid-plan.md b/.github/skills/internal-gateway-execute-plans/fixtures/valid-plan.md new file mode 100644 index 00000000..6bb7e03b --- /dev/null +++ b/.github/skills/internal-gateway-execute-plans/fixtures/valid-plan.md @@ -0,0 +1,22 @@ +## Goal + +Validate the plan execution CLI against a realistic plan shape. + +## Repository Preflight + +- **Target:** test fixture validation. +- **Anti-scope:** no runtime changes. +- **Validation Path:** `pytest -q tests/fixture/` +- **Stop Conditions:** fixture is incomplete. + +## Global Constraints + +- Preserve fixture integrity. + +## Task 1: Validate CLI + +**Files:** +- `scripts/plan_execution.py` + +- [ ] Run `python3 scripts/plan_execution.py preflight valid-plan.md` +- [ ] Confirm exit 0. diff --git a/.github/skills/internal-gateway-execute-plans/references/execution-contract.md b/.github/skills/internal-gateway-execute-plans/references/execution-contract.md new file mode 100644 index 00000000..a6558c34 --- /dev/null +++ b/.github/skills/internal-gateway-execute-plans/references/execution-contract.md @@ -0,0 +1,60 @@ +# Execution Contract + +This reference maps repository-local hooks around the delegated +`/superpowers-executing-plans` loop. It does not duplicate that skill's plan +review, todo, task execution, or core stop procedure. + +## Before the delegated loop + +- Require the exact retained plan path under `tmp/superpowers/plans/`. +- Confirm explicit user approval is present in the current conversation. +- Compute and record the SHA-256 plan fingerprint with `scripts/plan_execution.py`. +- Record branch, dirty files, and in-scope overlap before editing. +- Name the local task dependency set and focused validation command. +- Preserve the no-Git-mutation rule throughout execution. + +## Before each delegated task + +- State the task's observable outcome, dependency set, and focused validation. +- For executable or evaluable behavior, load `/internal-tdd` and establish + red-first evidence before the first implementation edit. +- Keep repository-owned routing, status, fixtures, and approval gates in scope; + do not edit imported core skills. + +## After each delegated task + +- Run the plan-specified focused validation command. +- Confirm the dependency set no longer asserts the replaced behavior. +- Retain fresh evidence before transitioning to the next task. +- If the plan authorizes simplification, `/addyosmani-code-simplification` + may be loaded at that task's explicit gate. + +## Execution discipline + +- **TF (tight feedback):** Within the current approved task, treat the smallest + coherent dependency set as one execution unit. Keep coupled edits, such as a + signature, its callers, and its implementation, together; keep unrelated plan + tasks separate. Use the task's focused validation as its transition gate. +- **FFD (fail fast on drift):** Stop the delegated loop when execution reveals + plan drift, an owner conflict, missing required validation, or unapproved + scope expansion. Record the gap and the exact evidence needed to continue in + the applicable status sibling. + +## Pause and resume + +- On pause, record the exact status sibling, remaining tasks, validation gap, + and next action using `references/status-contract.md`. +- On resume, run `resume-check` and require the recorded fingerprint to match + the retained plan before continuing. +- If the plan changed after approval, stop and record plan drift until the + approval and fingerprint are refreshed. + +## Closeout + +- Run all broader validation required by the retained plan. +- Load `/superpowers-verification-before-completion` before claiming completion. +- Run `git diff --check` and verify no Git mutation was performed. +- Replace older status siblings for the same plan basename and write exactly + one allowed status sibling. +- Run `completion-check` only when every task and broader check has fresh + passing evidence. diff --git a/.github/skills/internal-gateway-execute-plans/references/status-contract.md b/.github/skills/internal-gateway-execute-plans/references/status-contract.md new file mode 100644 index 00000000..495ccbc2 --- /dev/null +++ b/.github/skills/internal-gateway-execute-plans/references/status-contract.md @@ -0,0 +1,86 @@ +# Status Contract + +Status transition rules, required headings, and exact sibling filenames for plan closeout. + +## Status Transition Table + +| Current condition | Status | Continuation | +| --- | --- | --- | +| All tasks and broader validation have fresh passing evidence | `DONE` | none | +| At least one task is complete and executable tasks remain | `PARTIAL` | continuing | +| A named blocker prevents further execution | `BLOCKED` | waiting | +| Execution is complete but a human or external verification remains | `NEEDS_REVIEW` | waiting | + +Use `DONE` only when every task passed its transition gate, all in-scope work is complete, and required broader validation has fresh passing evidence. A final broad check does not retroactively validate skipped task gates. + +For any gap, use the status that best explains the remaining action and record the exact evidence needed to resume or finish. + +## Required Headings + +Every status file must contain these headings in order: + +```markdown +## Status +## Plan +## Plan Fingerprint +## Reason +## Workspace Baseline +## Files Changed +## Completed +## Remaining +## Validation +## Next +## Resume Notes +``` + +- **Status** — one of `DONE`, `PARTIAL`, `BLOCKED`, or `NEEDS_REVIEW`. +- **Plan** — the exact plan file path. +- **Plan Fingerprint** — the SHA-256 hash of the approved plan, prefixed with `sha256:`. +- **Reason** — why this status was chosen; for `BLOCKED`, name the blocker; for `DONE`, confirm all evidence is fresh. +- **Workspace Baseline** — branch, dirty files, and in-scope overlap at the time of closeout. +- **Files Changed** — list of files created or modified during execution. +- **Completed** — list of tasks that passed their transition gate, with task-level evidence. +- **Remaining** — list of tasks not yet complete, with the exact work remaining. +- **Validation** — list of validation commands run and their results. +- **Next** — the exact next action to resume or finish. +- **Resume Notes** — context needed to resume execution, including any drift or blockers. + +## Exact Allowed Sibling Filenames + +Status files use exact allowed sibling filenames: + +```text +.DONE.md +.PARTIAL.md +.BLOCKED.md +.NEEDS_REVIEW.md +``` + +Where `` is the plan filename without the `.md` extension. For example, if the plan is `2026-07-25-1831-self-contained-execute-plans.md`, the status file for `DONE` is `2026-07-25-1831-self-contained-execute-plans.DONE.md`. + +## Replacement Rules + +Before final response or pause: + +1. Identify the plan basename. +2. Check for existing status siblings with the same basename. +3. If an older sibling exists, replace it by writing the new status to a temporary sibling followed by an atomic rename. +4. Replacement considers only the four exact allowed names above; preserve every other sibling. +5. Write exactly one status sibling for the current closeout. + +## Resume Safety + +When resuming from a `PARTIAL` or `BLOCKED` status: + +1. Verify the status file exists and contains all required headings. +2. Compute the current plan fingerprint and compare it to the recorded `## Plan Fingerprint`. +3. If the fingerprints differ, the plan changed after approval; stop and record the drift. +4. If the fingerprints match, resume from the first task listed in `## Remaining`. +5. Do not resume from `DONE` unless fresh evidence invalidates the previous closeout. + +When resuming from `NEEDS_REVIEW`: + +1. Verify the status file exists and contains all required headings. +2. Confirm the human or external verification is complete. +3. If verification passed, update the status to `DONE` with fresh broader-validation evidence. +4. If verification failed, update the status to `BLOCKED` with the failure details. diff --git a/.github/skills/internal-gateway-execute-plans/scripts/plan_execution.py b/.github/skills/internal-gateway-execute-plans/scripts/plan_execution.py new file mode 100644 index 00000000..ba3925f6 --- /dev/null +++ b/.github/skills/internal-gateway-execute-plans/scripts/plan_execution.py @@ -0,0 +1,360 @@ +#!/usr/bin/env python3 +"""Read-only, stdlib-only plan execution validator CLI.""" +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import sys +from dataclasses import dataclass +from pathlib import Path +from typing import Literal + + +@dataclass(frozen=True) +class Finding: + code: str + message: str + severity: Literal["blocking", "notice"] = "blocking" + + +ALLOWED_STATUSES = frozenset({"DONE", "PARTIAL", "BLOCKED", "NEEDS_REVIEW"}) + +STATUS_FILENAME_RE = re.compile( + r"^(?P.+)\.(?PDONE|PARTIAL|BLOCKED|NEEDS_REVIEW)\.md$" +) + +REQUIRED_PLAN_HEADINGS = ( + "Goal", + "Repository Preflight", + "Global Constraints", +) + +REQUIRED_STATUS_HEADINGS = ( + "Status", + "Plan", + "Plan Fingerprint", + "Reason", + "Workspace Baseline", + "Files Changed", + "Completed", + "Remaining", + "Validation", + "Next", + "Resume Notes", +) + + +def compute_sha256(path: Path) -> str: + h = hashlib.sha256() + h.update(path.read_bytes()) + return f"sha256:{h.hexdigest()}" + + +def _extract_headings(text: str) -> list[str]: + return [ + line.lstrip("#").strip() + for line in text.splitlines() + if line.startswith("#") + ] + + +def _parse_status_from_filename(path: Path) -> str | None: + m = STATUS_FILENAME_RE.match(path.name) + if m: + return m.group("status") + return None + + +def _parse_status_from_content(text: str) -> str | None: + for line in text.splitlines(): + stripped = line.strip().strip("`") + if stripped in ALLOWED_STATUSES: + return stripped + return None + + +def validate_plan(path: Path, repo_root: Path) -> list[Finding]: + findings: list[Finding] = [] + if not path.is_file(): + findings.append(Finding("plan-not-found", f"Plan file not found: {path}")) + return findings + + retained_dir = repo_root / "tmp" / "superpowers" / "plans" + try: + path.resolve().relative_to(retained_dir.resolve()) + except ValueError: + findings.append( + Finding( + "plan-outside-retained-directory", + f"Plan must be under {retained_dir}", + ) + ) + + text = path.read_text() + headings = _extract_headings(text) + heading_set = set(headings) + + for required in REQUIRED_PLAN_HEADINGS: + if required not in heading_set: + inline_pattern = f"**{required}:**" + if inline_pattern not in text: + findings.append( + Finding("missing-heading", f"Plan missing required heading: {required}") + ) + + if "Task" not in " ".join(headings) and "## Task" not in text: + findings.append( + Finding("missing-task", "Plan must contain at least one task heading") + ) + + return findings + + +def validate_status(path: Path) -> list[Finding]: + findings: list[Finding] = [] + if not path.is_file(): + findings.append( + Finding("status-not-found", f"Status file not found: {path}") + ) + return findings + + text = path.read_text() + headings = _extract_headings(text) + heading_set = set(headings) + + status_from_file = _parse_status_from_filename(path) + status_from_content = _parse_status_from_content(text) + + if status_from_file is None: + findings.append( + Finding( + "unknown-status", + f"Status filename must match ..md with STATUS in {sorted(ALLOWED_STATUSES)}", + ) + ) + elif status_from_content is not None and status_from_file != status_from_content: + findings.append( + Finding( + "status-mismatch", + f"Filename status {status_from_file} != content status {status_from_content}", + ) + ) + + for required in REQUIRED_STATUS_HEADINGS: + if required not in heading_set: + findings.append( + Finding( + "missing-heading", + f"Status missing required heading: {required}", + ) + ) + + return findings + + +def validate_resume(plan_path: Path, status_path: Path, repo_root: Path | None = None) -> list[Finding]: + findings: list[Finding] = [] + effective_root = repo_root or _find_repo_root(plan_path) + findings.extend(validate_plan(plan_path, repo_root=effective_root)) + findings.extend(validate_status(status_path)) + + if any(f.severity == "blocking" for f in findings): + return findings + + plan_fingerprint = compute_sha256(plan_path) + status_text = status_path.read_text() + + recorded_fingerprint = None + for line in status_text.splitlines(): + stripped = line.strip().strip("`") + if stripped.startswith("sha256:"): + recorded_fingerprint = stripped + break + + if recorded_fingerprint is None: + findings.append( + Finding( + "missing-fingerprint", + "Status file must contain a Plan Fingerprint heading with sha256: value", + ) + ) + elif recorded_fingerprint != plan_fingerprint: + findings.append( + Finding( + "plan-fingerprint-drift", + f"Plan changed after approval: recorded {recorded_fingerprint} != computed {plan_fingerprint}", + ) + ) + + return findings + + +def validate_completion(plan_path: Path, status_path: Path, repo_root: Path | None = None) -> list[Finding]: + findings: list[Finding] = [] + findings.extend(validate_resume(plan_path, status_path, repo_root=repo_root)) + + if any(f.severity == "blocking" for f in findings): + return findings + + status_from_file = _parse_status_from_filename(status_path) + if status_from_file != "DONE": + findings.append( + Finding( + "not-done", + f"Completion requires DONE status, got {status_from_file}", + ) + ) + + status_text = status_path.read_text() + headings = _extract_headings(status_text) + if "Remaining" in headings: + idx = headings.index("Remaining") + lines = status_text.splitlines() + remaining_lines = [] + collecting = False + for line in lines: + if line.strip() == "## Remaining": + collecting = True + continue + if collecting and line.startswith("## "): + break + if collecting and line.strip(): + remaining_lines.append(line.strip()) + + has_real_items = any( + item for item in remaining_lines if item.lower() not in ("none", "- none") + ) + if has_real_items: + findings.append( + Finding( + "remaining-items", + "DONE status requires no remaining items", + ) + ) + + return findings + + +def build_compact_payload(findings: list[Finding]) -> dict[str, object]: + blocking = [f for f in findings if f.severity == "blocking"] + notice = [f for f in findings if f.severity == "notice"] + sample = [ + {"code": f.code, "severity": f.severity} for f in findings[:10] + ] + return { + "status": "passed" if not blocking else "failed", + "finding_counts": { + "total": len(findings), + "blocking": len(blocking), + "notice": len(notice), + }, + "finding_sample": sample, + "next_action": ( + "All checks passed." + if not blocking + else "Resolve blocking plan execution findings." + ), + } + + +def _format_text(findings: list[Finding]) -> str: + if not findings: + return "OK: all checks passed." + lines = [] + for f in findings: + lines.append(f"[{f.severity.upper()}] {f.code}: {f.message}") + return "\n".join(lines) + + +def _format_json(findings: list[Finding]) -> str: + return json.dumps( + { + "status": "passed" if not findings else "failed", + "findings": [ + {"code": f.code, "message": f.message, "severity": f.severity} + for f in findings + ], + }, + indent=2, + ) + + +def _find_repo_root(start: Path) -> Path: + for parent in start.resolve().parents: + if (parent / "AGENTS.md").exists() and (parent / ".github").exists(): + return parent + return start.resolve() + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + description="Read-only plan execution validator" + ) + subparsers = parser.add_subparsers(dest="command", required=True) + + preflight = subparsers.add_parser("preflight", help="Validate a plan file") + preflight.add_argument("plan", type=Path) + preflight.add_argument("--repo-root", type=Path, default=None) + preflight.add_argument( + "--format", choices=("text", "json", "compact"), default="text" + ) + + status_check = subparsers.add_parser("status-check", help="Validate a status file") + status_check.add_argument("status", type=Path) + status_check.add_argument( + "--format", choices=("text", "json", "compact"), default="text" + ) + + resume_check = subparsers.add_parser( + "resume-check", help="Validate resume safety" + ) + resume_check.add_argument("plan", type=Path) + resume_check.add_argument("status", type=Path) + resume_check.add_argument("--repo-root", type=Path, default=None) + resume_check.add_argument( + "--format", choices=("text", "json", "compact"), default="text" + ) + + completion_check = subparsers.add_parser( + "completion-check", help="Validate completion readiness" + ) + completion_check.add_argument("plan", type=Path) + completion_check.add_argument("status", type=Path) + completion_check.add_argument("--repo-root", type=Path, default=None) + completion_check.add_argument( + "--format", choices=("text", "json", "compact"), default="text" + ) + + args = parser.parse_args(argv) + + if args.command == "preflight": + repo_root = args.repo_root or _find_repo_root(args.plan) + findings = validate_plan(args.plan, repo_root=repo_root) + elif args.command == "status-check": + findings = validate_status(args.status) + elif args.command == "resume-check": + findings = validate_resume(args.plan, args.status, repo_root=args.repo_root) + elif args.command == "completion-check": + findings = validate_completion(args.plan, args.status, repo_root=args.repo_root) + else: + parser.print_help(sys.stderr) + return 2 + + fmt = getattr(args, "format", "text") + if fmt == "compact": + output = json.dumps(build_compact_payload(findings)) + elif fmt == "json": + output = _format_json(findings) + else: + output = _format_text(findings) + + sys.stdout.write(output + "\n") + + blocking = any(f.severity == "blocking" for f in findings) + return 1 if blocking else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/skills/internal-gateway-idea/SKILL.md b/.github/skills/internal-gateway-idea/SKILL.md index a8dc4f18..79be54f0 100644 --- a/.github/skills/internal-gateway-idea/SKILL.md +++ b/.github/skills/internal-gateway-idea/SKILL.md @@ -1,23 +1,25 @@ --- name: internal-gateway-idea description: Use when a repository-owned idea needs brainstorming, assumption challenge, alternative discovery, and a spec-vs-plan recommendation before implementation planning. +disable-model-invocation: true --- # Internal Gateway Idea ## Referenced skills -- `superpowers-brainstorming`: core idea-to-design workflow. -- `internal-gateway-writing-plans`: retained spec or implementation-plan writing after the user approves the direction. -- `mattpocock-research`: on-demand owner for decision-relevant external research after local evidence is exhausted; this reference does not preload the skill. +- `/superpowers-brainstorming`: core idea-to-design workflow; retained spec writing and review stay with this owner. +- `/internal-gateway-writing-plans`: implementation-plan writing only, after the user approves the direction. +- `/mattpocock-research`: on-demand owner for decision-relevant external research after local evidence is exhausted; this reference does not preload the skill. ## Local references - `references/workflow.md`: authoritative state machine, Mermaid workflow, approval rules, routing stability, and scoped local validation lane for this bundle. -- `scripts/audit_workflow.py`: marker-consistency validator; run via `python3 scripts/audit_workflow.py` or `make internal-gateway-idea-fast-check`. +- `scripts/audit_workflow.py`: marker-consistency validator; run via `python3 .github/skills/internal-gateway-idea/scripts/audit_workflow.py` or `make internal-gateway-idea-fast-check`. +- Script output contract: `text` for short operator summaries (default), `json` for nested or machine-consumed output, `tsv`/`csv` only for large flat tables; data on stdout, diagnostics on stderr; keep output bounded. -Lightweight repository-owned wrapper for idea shaping. Use `superpowers-brainstorming` as the core workflow and add the local gates below. Loading `superpowers-brainstorming` is an intentional, globally-resolvable exception to the bundle self-containment rule. This skill does not replace the core brainstorming process; it constrains it for repository-owned idea work. +Lightweight repository-owned wrapper for idea shaping. Use `/superpowers-brainstorming` as the core workflow and add the local gates below. Loading `/superpowers-brainstorming` is an intentional, globally-resolvable exception to the bundle self-containment rule. This skill does not replace the core brainstorming process; it constrains it for repository-owned idea work. ## When to use @@ -35,21 +37,47 @@ Lightweight repository-owned wrapper for idea shaping. Use `superpowers-brainsto ## Core contract - Follow the mandatory gate sequence: `Specialization Checkpoint: gated`, `Idea Gate 0`, `External Research Checkpoint`, `Assumption Challenge Gate`, `Alternative discovery`, `Critical Challenge Gate`, `Spec vs plan decision`, `Stop before implementation execution`. -- Load `superpowers-brainstorming` as the core workflow. +- Load `/superpowers-brainstorming` as the core workflow. - Read `references/workflow.md` before presenting the final design direction. -- Keep the `superpowers-brainstorming` hard gate: no implementation action before the user approves the design or direct-plan recommendation. +- Keep the `/superpowers-brainstorming` hard gate: no implementation action before the user approves the design or direct-plan recommendation. - Treat approval as gate-local. `procedi`, `ok`, `go`, or similar approval advances only the active visible gate. - If approval wording is ambiguous, ask whether it means critical review, retained spec or plan writing, or implementation execution. -- After the bounded evidence pass, run `Idea Gate 0` as a visible numbered question block with `Question`, `Recommendation`, `Why`, and `Default if accepted`; evidence cannot replace Idea Gate 0. +- After the bounded evidence pass, run `Idea Gate 0` as one numbered bulk question block; evidence cannot replace Idea Gate 0. - Do not proceed to `External Research Checkpoint`, assumption challenge, alternative discovery, design direction, critical challenge, or spec-vs-plan decision until `Idea Gate 0` is accepted or the user explicitly overrides its defaults. - Run `Critical Challenge Gate` as its own visible gate after the user approves the design direction and before the spec-vs-plan decision; an embedded critique does not satisfy Critical Challenge Gate. - If any mandatory gate was skipped, stop, name the missed gate, mark any downstream artifact as draft-only, and resume at the first skipped mandatory gate. - Use this skill only to add repository-owned idea gates, not to fork the core brainstorming process. - Keep collaborative questioning inside the core brainstorming workflow. -- Load `internal-gateway-writing-plans` only after the user approves retained spec or implementation-plan writing. +- Load `/internal-gateway-writing-plans` only after the user approves implementation-plan writing, either directly or after retained-spec review. - Stop after the delegated writing outcome. Do not implement, invoke execution owners, or run execution commands from this skill. - Keep the agent filename, frontmatter name, and workflow aligned. +## User-facing communication + +Keep mandatory gates, research decisions, assumption checks, approval state, +and recovery state as internal workflow state. Project only decision-relevant +information into chat through one compact user-facing decision card for status, +approval, routing, material risk, blocker, and requested user action. + +Use at most four content lines: + +- `🧭` names one unresolved decision. +- `✅` gives the recommendation or result. +- `💡` gives one short reason when it changes the decision. +- `✈️` states the exact user action and what acceptance advances. + +Use `🎯` for a goal, `🛠️` for a proposed change, `🧪` for validation, and +`⚠️` for a material risk or blocker when one of those facts is the decision. + +Content-bearing output uses its owning schema and is outside the four-line +card limit. Guided questions, alternatives, design sections, and required +critique schemas are content-bearing output. + +Do not announce skipped checkpoints. Do not print the internal gate ledger, +bounded-evidence notes, anti-scope inventory, research checkpoint, or routing +bookkeeping unless the user asks for details or one item blocks progress. +Match the user's language. + ## Bounded context pass Before asking the first question block: @@ -81,7 +109,7 @@ and resume there before producing or revising a retained artifact. Do not skip from evidence, design approval, or `Decision: direct plan` to implementation. The only post-brainstorming owner this skill may load is -`internal-gateway-writing-plans`, and only after explicit approval for the +`/internal-gateway-writing-plans`, and only after explicit approval for the selected writing path. ## Idea Gate 0 @@ -89,17 +117,20 @@ selected writing path. Run this gate after bounded evidence and before any challenge or alternative recommendation. -Use one visible numbered question block. Each question must include: +Render all currently known unresolved questions in one numbered bulk question +block. Each question must include: -- `Question` -- `Recommendation` -- `Why` -- `Default if accepted` +- `Question:` the unresolved decision. +- `Recommendation:` the evidence-based default. +- `Why:` one evidence-based sentence. +- `Default if accepted:` the concrete consequence of acceptance. -Questions should focus only on decisions repository evidence cannot safely -answer: intent, accepted defaults, constraints, success criteria, validation -path, and anti-scope. A bounded evidence pass may prepare recommended defaults, -but evidence cannot replace Idea Gate 0. +Capture the question, recommendation, reason, and accepted default internally. +Evidence-based defaults require explicit acceptance. Questions should focus +only on decisions repository evidence cannot safely answer: intent, accepted +defaults, constraints, success criteria, validation path, and anti-scope. A +bounded evidence pass may prepare recommended defaults, but evidence cannot +replace Idea Gate 0. ## External Research Checkpoint @@ -115,7 +146,7 @@ Skip external research unless all of these are true: When all conditions hold: 1. Define one bounded research question. -2. load `mattpocock-research` on-demand and write one Markdown report under +2. load `/mattpocock-research` on-demand and write one Markdown report under `tmp/research/YYYY-MM-DD-.md`. 3. Bring only the report path and decision-relevant conclusions back into the brainstorming flow. @@ -123,7 +154,7 @@ When all conditions hold: evidence changes an accepted constraint or default. `internal-gateway-idea` owns when research is warranted. -`mattpocock-research` owns how the research is performed. Do not copy its +`/mattpocock-research` owns how the research is performed. Do not copy its research procedure here, and do not start a second research pass automatically. Validation must keep this checkpoint on-demand, bounded to one question and one @@ -182,6 +213,8 @@ Keep this gate inside the local idea wrapper and the core brainstorming workflow After the user approves the design direction, decide whether to go straight to an implementation plan or write a retained spec first. +Retained spec writing stays with `/superpowers-brainstorming`. After the spec is written and reviewed, the user may then approve implementation-plan writing, which loads `/internal-gateway-writing-plans`. + Choose `Decision: direct plan` when target, owner, scope, constraints, rejected alternatives, and validation path are clear enough that a retained spec would mostly duplicate the implementation plan. Choose `Decision: spec first` when product, architecture, data flow, rollout, ownership, or risk decisions are still material enough that a retained spec would reduce the chance of building the wrong thing. @@ -191,23 +224,22 @@ Always explain the choice: - `Decision: direct plan` or `Decision: spec first` - `Why:` one evidence-based sentence. - `Rejected option:` the other path and why it is weaker. -- `Next owner:` `internal-gateway-writing-plans` after user approval. -- `Approval request:` ask the user to approve the selected writing path before loading the next owner. +- `Next owner:` `/internal-gateway-writing-plans` after user approval for implementation-plan writing. +- `Approval request:` ask the user to approve the selected path before loading the next owner. -Approval of `Decision: direct plan` skips a retained spec only. It does not -authorize implementation execution. +Approval of `Decision: direct plan` skips a retained spec only. It does not authorize implementation execution. ## Validation - The skill read `references/workflow.md` before finalizing the design direction. - The skill used `Specialization Checkpoint: gated` for execution-shaped requests. -- The skill loaded `superpowers-brainstorming` as core instead of copying its workflow. +- The skill loaded `/superpowers-brainstorming` as core instead of copying its workflow. - The skill challenged the user's initial assumption, not only corrected the proposed solution. - The skill presented 2-3 approaches and explained why the recommendation beat the strongest rejected option. - The skill used `Critical Challenge Gate` before spec or plan writing. - The skill treated ambiguous approval words as gate-local and clarified the active gate when needed. - The skill kept collaborative questioning inside the core brainstorming workflow. -- The skill loaded `internal-gateway-writing-plans` only for retained spec or implementation-plan writing. -- The skill stopped after `internal-gateway-writing-plans` produced a writing outcome. +- The skill loaded `/internal-gateway-writing-plans` only for implementation-plan writing after user approval. +- The skill stopped after `/internal-gateway-writing-plans` produced a writing outcome. - The Mermaid workflow and runtime prompt contain the same mandatory gate names. - The phrase `agent filename, frontmatter name, and workflow aligned` appears in the skill, workflow, and runtime prompt. diff --git a/.github/skills/internal-gateway-idea/agents/openai.yaml b/.github/skills/internal-gateway-idea/agents/openai.yaml index b2bef3d3..da4303b8 100644 --- a/.github/skills/internal-gateway-idea/agents/openai.yaml +++ b/.github/skills/internal-gateway-idea/agents/openai.yaml @@ -2,21 +2,33 @@ interface: display_name: "Internal Gateway Idea" short_description: "Idea brainstorming with assumption challenge" default_prompt: >- - Use $internal-gateway-idea when a repository-owned idea, proposed direction, + Use /internal-gateway-idea when a repository-owned idea, proposed direction, unclear goal, or option set needs brainstorming before planning. Load - $superpowers-brainstorming as the core workflow, then read + /superpowers-brainstorming as the core workflow, then read references/workflow.md for the local state machine. If the incoming request - is execution-shaped, emit Specialization Checkpoint: gated and continue - through Idea Gate 0 before any recommendation. Run Idea Gate 0 as a visible - numbered question block; evidence cannot replace Idea Gate 0. Add the local + is execution-shaped, record Specialization Checkpoint: gated internally and + project only the decision-relevant consequence, then continue through Idea + Gate 0 before any recommendation. Run Idea Gate 0 as one numbered bulk question block with Question, Recommendation, Why, and Default if accepted; + evidence cannot replace Idea Gate 0. Follow the mandatory gate sequence: + Specialization Checkpoint: gated, Idea Gate 0, External Research Checkpoint, Assumption Challenge Gate, Alternative discovery, Critical Challenge Gate, - and visible spec-vs-plan decision from SKILL.md. Run Critical Challenge Gate - as its own visible gate after design-direction approval; an embedded critique - does not satisfy Critical Challenge Gate. Treat procedi, ok, go, or similar - approval as advancing only the active visible gate; clarify ambiguous - approval before loading another owner. If a mandatory gate was skipped, stop - and resume at the first skipped mandatory gate. Load - $internal-gateway-writing-plans only after the user explicitly approves - retained spec or implementation-plan writing. Stop after the writing outcome; do not implement or invoke execution owners from this skill. Keep the agent filename, frontmatter name, and workflow aligned. -name: internal-gateway-idea -description: Use when a repository-owned idea needs brainstorming, assumption challenge, alternative discovery, and a spec-vs-plan recommendation before implementation planning. + Spec vs plan decision, Stop before implementation execution. External + research is on-demand after local evidence is insufficient; load + /mattpocock-research only when one external fact could change feasibility, + approach, constraints, or risk. Run Critical Challenge Gate as its own + visible gate after design-direction approval; an embedded critique does not + satisfy Critical Challenge Gate. Treat procedi, ok, go, or similar approval + as advancing only the active visible gate; clarify ambiguous approval before + loading another owner. If a mandatory gate was skipped, stop and resume at + the first skipped mandatory gate. Load /internal-gateway-writing-plans only + after the user explicitly approves implementation-plan writing. Retained spec + writing stays with /superpowers-brainstorming. Stop after the writing outcome; + do not implement or invoke execution owners from this skill. + Keep mandatory gates as internal workflow state. In chat, emit one compact user-facing decision card with 🎯 🧭 🛠️ 🧪 ⚠️ ✅ 💡 ✈️ for status, approval, + routing, material risk, blocker, and requested user action. Content-bearing + output uses its owning schema and is outside the four-line card limit. + Guided questions, alternatives, design sections, and required critique + schemas are content-bearing output. Do + not announce skipped checkpoints or print internal ledgers. + Always show a material risk, blocker, validation gap, or user decision. + Keep the agent filename, frontmatter name, and workflow aligned. diff --git a/.github/skills/internal-gateway-idea/references/workflow.md b/.github/skills/internal-gateway-idea/references/workflow.md index a21566e1..209e1a38 100644 --- a/.github/skills/internal-gateway-idea/references/workflow.md +++ b/.github/skills/internal-gateway-idea/references/workflow.md @@ -1,6 +1,8 @@ # Internal Gateway Idea Workflow This workflow defines the canonical contract for `internal-gateway-idea`. +The delegated core workflow is `/superpowers-brainstorming`; retained-spec +writing remains with that owner. ## State Machine @@ -18,7 +20,7 @@ flowchart TD E -- no --> D E -- yes --> F{External Research Checkpoint} F -- skip --> H[Assumption Challenge Gate] - F -- research needed --> G[Load mattpocock-research on-demand] + F -- research needed --> G[Load /mattpocock-research on-demand] G --> G1[Write one report under tmp/research/] G1 --> G2{Accepted defaults changed?} G2 -- yes --> D @@ -33,13 +35,15 @@ flowchart TD M -- narrow --> J M -- continue --> N[Spec vs plan decision] N --> O{Decision} - O -- spec first --> P[Ask approval for retained spec path] - O -- direct plan --> Q[Ask approval for direct plan path] - P --> R{Approved?} - Q --> R - R -- no --> N - R -- yes --> S[Load internal-gateway-writing-plans] - S --> T[Writing outcome only] + O -- spec first --> P1[Write retained spec in brainstorming] + P1 --> P2[Self-review] + P2 --> P3[User reviews retained spec] + P3 --> P4[Approve implementation-plan writing] + P4 --> P5[Load /internal-gateway-writing-plans] + O -- direct plan --> Q1[Approve implementation-plan writing] + Q1 --> Q2[Load /internal-gateway-writing-plans] + P5 --> T[Writing outcome only] + Q2 --> T T --> U[Stop before implementation execution] ``` @@ -49,13 +53,13 @@ flowchart TD | --- | --- | --- | | `Specialization Checkpoint: gated` | Use when the incoming ask is already a file edit, command run, validator run, implementation step, or other execution-shaped request. Name the later execution owner only as a future consequence. | Do not execute, hand off, or present the post-critical recommendation. | | `Skipped-gate recovery` | Stop the current lane, name the first skipped mandatory gate, mark downstream artifacts draft-only, and resume at that gate. | Do not continue from an invalid later state or ask the user to approve a handoff built on skipped gates. | -| `Idea Gate 0` | Confirm the recovered intent, defaults, constraints, success criteria, validation path, and anti-scope with a visible numbered question block using `Question`, `Recommendation`, `Why`, and `Default if accepted`; evidence cannot replace Idea Gate 0. | Do not treat repository evidence alone as user approval, and do not proceed to challenge, alternatives, design, or planning until this gate is accepted. | -| `External Research Checkpoint` | Skip unless local evidence is insufficient and one external fact could change feasibility, approach, constraints, or risk. When needed, load `mattpocock-research` on-demand with one bounded question, write one Markdown report under `tmp/research/`, and return only decision-relevant conclusions. | Do not preload the research skill, copy its research procedure, run generic best-practice research, or start a second research pass automatically. | +| `Idea Gate 0` | Confirm the recovered intent, defaults, constraints, success criteria, validation path, and anti-scope internally; render all currently known unresolved questions in one numbered bulk question block with `Question`, `Recommendation`, `Why`, and `Default if accepted`; evidence cannot replace Idea Gate 0. | Do not treat repository evidence alone as user approval, and do not proceed to challenge, alternatives, design, or planning until this gate is accepted. | +| `External Research Checkpoint` | Skip unless local evidence is insufficient and one external fact could change feasibility, approach, constraints, or risk. When needed, load `/mattpocock-research` on-demand with one bounded question, write one Markdown report under `tmp/research/`, and return only decision-relevant conclusions. | Do not preload the research skill, copy its research procedure, run generic best-practice research, or start a second research pass automatically. | | `Assumption Challenge Gate` | Test whether the proposed target or solution is necessary before choosing an approach. | Do not only polish the user's proposed solution. | | `Alternative discovery` | Present 2-3 approaches and explain why the recommended one beats the strongest rejected option. | Do not present a single-path design as inevitable. | -| `Critical Challenge Gate` | Challenge the chosen direction as its own visible gate after design-direction approval and before spec or plan writing. Reopen or narrow when the objection is material. | Do not use this gate after loading `internal-gateway-writing-plans`; an embedded critique does not satisfy Critical Challenge Gate. | -| `Spec vs plan decision` | Choose `Decision: direct plan` or `Decision: spec first`, explain why, name the rejected path, and ask for approval. | Do not load `internal-gateway-writing-plans` from the decision alone. | -| `Writing outcome only` | Load `internal-gateway-writing-plans` only after explicit user approval for the selected writing path. Stop after the delegated writing outcome. | Do not implement, edit target files, run execution commands, or invoke execution owners. | +| `Critical Challenge Gate` | Challenge the chosen direction as its own visible gate after design-direction approval and before spec or plan writing. Reopen or narrow when the objection is material. | Do not use this gate after loading `/internal-gateway-writing-plans`; an embedded critique does not satisfy Critical Challenge Gate. | +| `Spec vs plan decision` | Choose `Decision: direct plan` or `Decision: spec first`, explain why, name the rejected path, and ask for approval. | Do not load `/internal-gateway-writing-plans` from the decision alone. | +| `Writing outcome only` | Load `/internal-gateway-writing-plans` only after explicit user approval for the selected writing path. Stop after the delegated writing outcome. | Do not implement, edit target files, run execution commands, or invoke execution owners. | ## Approval Rules @@ -72,8 +76,14 @@ flowchart TD Keep the agent filename, frontmatter name, and workflow aligned. +The state-machine labels are internal workflow state. Normal chat emits one compact user-facing decision card for status, approval, routing, material risk, blocker, and requested user action, and never dumps the state-machine trace. +Content-bearing output uses its owning schema and is outside the four-line card limit. Guided questions, alternatives, design sections, and required critique schemas are content-bearing output. +Use 🎯 for a goal, 🧭 for a decision, 🛠️ for a proposed change, 🧪 for validation, ⚠️ for a material risk or blocker, ✅ for a recommendation or result, 💡 for a short reason, and ✈️ for a requested user action. +Skipped checkpoints remain silent. Material risks, blockers, validation gaps, +and user decisions remain visible. + ## Local validation lane -Run `python3 scripts/audit_workflow.py` or `make internal-gateway-idea-fast-check` before widening to catalog-wide checks. This scoped lane must cover the bundle audit and marker consistency. +Run `python3 .github/skills/internal-gateway-idea/scripts/audit_workflow.py` or `make internal-gateway-idea-fast-check` before widening to catalog-wide checks. This scoped lane must cover the bundle audit and marker consistency. The checkpoint must use one bounded research question; do not start a second research pass automatically. diff --git a/.github/skills/internal-gateway-idea/scripts/audit_workflow.py b/.github/skills/internal-gateway-idea/scripts/audit_workflow.py index a8f37208..14aebf65 100644 --- a/.github/skills/internal-gateway-idea/scripts/audit_workflow.py +++ b/.github/skills/internal-gateway-idea/scripts/audit_workflow.py @@ -22,6 +22,7 @@ def contains_in_order(text: str, markers: list[str]) -> bool: MANDATORY_SEQUENCE = [ "Specialization Checkpoint: gated", "Idea Gate 0", + "External Research Checkpoint", "Assumption Challenge Gate", "Alternative discovery", "Critical Challenge Gate", @@ -32,8 +33,12 @@ def contains_in_order(text: str, markers: list[str]) -> bool: RUNTIME_SEQUENCE = [ "Specialization Checkpoint: gated", "Idea Gate 0", + "External Research Checkpoint", + "Assumption Challenge Gate", + "Alternative discovery", "Critical Challenge Gate", - "spec-vs-plan decision", + "Spec vs plan decision", + "Stop before implementation execution", ] @@ -46,7 +51,7 @@ def main() -> int: shared_markers = [ "Specialization Checkpoint: gated", "Idea Gate 0", - "visible numbered question block", + "compact user-facing decision card", "Assumption Challenge Gate", "Alternative discovery", "Critical Challenge Gate", @@ -55,6 +60,18 @@ def main() -> int: "internal-gateway-writing-plans", "Stop before implementation execution", ] + chat_projection_markers = [ + "compact user-facing decision card", + "internal workflow state", + "🎯", + "🧭", + "🛠️", + "🧪", + "⚠️", + "✅", + "💡", + "✈️", + ] workflow_only_markers = [ "flowchart TD", "Approval Rules", @@ -73,9 +90,9 @@ def main() -> int: "do not start a second research pass automatically", ] runtime_only_markers = [ - "$internal-gateway-idea", - "$superpowers-brainstorming", - "$internal-gateway-writing-plans", + "/internal-gateway-idea", + "/superpowers-brainstorming", + "/internal-gateway-writing-plans", "do not implement", "agent filename, frontmatter name, and workflow aligned", ] @@ -90,14 +107,22 @@ def main() -> int: "Specialization Checkpoint: gated", "Idea Gate 0", "Critical Challenge Gate", - "spec-vs-plan decision", + "Spec vs plan decision", ], ), "skill_gate_sequence": contains_in_order(skill_text, MANDATORY_SEQUENCE), "workflow_gate_sequence": contains_in_order(workflow_text, MANDATORY_SEQUENCE), "runtime_gate_sequence": contains_in_order(runtime_text, RUNTIME_SEQUENCE), - "local_fast_lane_documented": "scripts/audit_workflow.py" in skill_text - and "scripts/audit_workflow.py" in workflow_text, + "runtime_research_checkpoint": contains_all( + runtime_text, + [ + "External Research Checkpoint", + "mattpocock-research", + "on-demand", + ], + ), + "local_fast_lane_documented": ".github/skills/internal-gateway-idea/scripts/audit_workflow.py" in skill_text + and ".github/skills/internal-gateway-idea/scripts/audit_workflow.py" in workflow_text, "workflow_mermaid_and_rules": contains_all(workflow_text, workflow_only_markers), "skill_research_escalation": contains_all(skill_text, research_markers), "workflow_research_escalation": contains_all( @@ -119,6 +144,19 @@ def main() -> int: "stop_before_execution": "Stop before implementation execution" in workflow_text and "Stop after the delegated writing outcome" in skill_text and "Stop after the writing outcome" in runtime_text, + "compact_chat_projection": all( + contains_all(text, chat_projection_markers) + for text in (skill_text, workflow_text, runtime_text) + ), + "retained_spec_owner": "Retained spec writing stays with `/superpowers-brainstorming`" in skill_text, + "spec_review_before_plan_handoff": "Write retained spec" in workflow_text + and "User reviews retained spec" in workflow_text + and "Approve implementation-plan writing" in workflow_text + and "Load /internal-gateway-writing-plans" in workflow_text, + "writing_gateway_is_plan_only": "implementation-plan writing" in skill_text + and "retained spec or implementation-plan writing" not in skill_text + and "retained spec or implementation-plan writing" not in workflow_text + and "retained spec or implementation-plan writing" not in runtime_text, } payload = {"strict_ok": all(markers.values()), "markers": markers} print(json.dumps(payload, indent=2, sort_keys=True)) diff --git a/.github/skills/internal-gateway-simple-task/SKILL.md b/.github/skills/internal-gateway-simple-task/SKILL.md index 68c244ec..5dc071eb 100644 --- a/.github/skills/internal-gateway-simple-task/SKILL.md +++ b/.github/skills/internal-gateway-simple-task/SKILL.md @@ -7,18 +7,31 @@ description: Use when a concrete low-to-medium-risk repository-owned coding or n ## Referenced skills -- `grill-me`: compact Gate 1 interview after local preflight and Initial Idea Ordering. -- `internal-gateway-critical-master`: Gate 2 challenge before non-trivial action. -- `internal-tdd`: executable or evaluable behavior changes that need repository-owned TDD routing before implementation. -- `superpowers-verification-before-completion`: final evidence gate before completion, readiness, passing, fixed, or no-gap claims. -- `addyosmani-code-simplification`: on-demand method owner only for an explicit code-simplification request or an already-approved simplification remediation. +- `/grill-me`: compact Gate 1 interview after local preflight and Initial Idea Ordering. +- `/internal-gateway-critical-master`: Gate 2 challenge before non-trivial action. +- `/internal-tdd`: executable or evaluable behavior changes that need repository-owned TDD routing before implementation. +- `/superpowers-verification-before-completion`: final evidence gate before completion, readiness, passing, fixed, or no-gap claims. +- `/addyosmani-code-simplification`: on-demand method owner only for an explicit code-simplification request or an already-approved simplification remediation. Do not introduce other skills, agents, or workflow owners from this bundle. When stopping, explain the violated condition and let the user choose the next path. +## Local references + +- Read `references/simple-lanes.md` after gate classification when the + `answer`, `edit`, `diagnose`, or `validate` posture is not already obvious. +- Read `references/clarification-gate.md` only when one missing bounded fact + blocks the active lane. +- Read `references/plan-mode.md` only when cost or complexity may require a + retained plan. +- Read `references/support-routing.md` only when the next method or evidence + requirement remains noisy after lane selection. + ## Core Contract Use this skill as the fast path for concrete bounded work that should finish in the current run. It owns the decision to continue or stop, then answers, edits, diagnoses, or validates end-to-end when the target, anti-scope, and validation path are concrete enough to execute safely. +Ordinary `full-gate` work continues after the internal Readiness Brief without redundant approval. An approval boundary produces `stop-with-reason`. Non-trivial work with an established validation gap stops before execution. `trivial-skip` may report an exact bounded validation gap. `gate requirements` are pre-action expectations; Gate Evidence is the actual internal ledger recorded during work. The default helper text has no more than four content lines. + Stop only when the work becomes materially complex, too costly for the current run, ambiguous, unsafe, multi-phase, approval-bound, or not locally verifiable. Stop output must explain the boundary break instead of delegating by name. Use local references and scripts only to keep the decision process compact and deterministic. Keep working state small, inspect bounded evidence first, and preserve direct completion ownership inside this bundle. @@ -61,11 +74,11 @@ For `full-gate`, complete the gates in this order: 1. Inspect the nearest local evidence. 2. Confirm the task still fits one bounded run. 3. Complete Initial Idea Ordering. -4. Ask one compact `grill-me` block only when a missing bounded fact blocks the active lane. -5. Run the critical challenge before non-trivial action. +4. Ask one compact `/grill-me` block only when a missing bounded fact blocks the active lane. Use `/grill-me` only when a missing bounded fact blocks the active lane. +5. Run `/internal-gateway-critical-master` before non-trivial action. 6. Write a short Readiness Brief. -7. If executable or evaluable behavior changes, load `internal-tdd` before implementation and follow its routed posture. -8. For an explicit code-simplification request or already-approved simplification remediation, establish a passing behavior baseline, then load `addyosmani-code-simplification`; do not create a simplification pass after unrelated implementation. +7. If executable or evaluable behavior changes, load `/internal-tdd` before implementation and follow its routed posture. +8. For an explicit code-simplification request or already-approved simplification remediation, establish a passing behavior baseline, then load `/addyosmani-code-simplification`; do not create a simplification pass after unrelated implementation. 9. Execute the smallest coherent in-scope move. 10. Run focused validation or report the exact validation gap. 11. Use the final evidence gate before positive claims. @@ -133,11 +146,32 @@ Keep the implementation posture simple: - Give each changed section, helper, or code path one current reason to exist. - When behavior changes, name the observable contract and validate it with the closest stable check. +## User-facing communication + +Keep the Readiness Brief fields as an internal readiness record. Keep Gate +Evidence internal by default; normal chat must not dump either structure. + +Render one compact user-facing projection: + +- `🧭` names the quick plan or decision. +- `🎯` states the goal. +- `🛠️` states the bounded change. +- `🧪` states the validation path or exact gap. +- `⚠️` states a material risk or blocker when present. +- `✅` states a verified result. +- `💡` gives one short reason when it changes the decision. +- `✈️` states the exact user action when approval or direction is required. + +Omit fields that do not affect the user. Use no more than four content lines. +Use `--format json` when complete readiness and evidence data is required. + ## Deterministic Helpers -- `scripts/resolve_simple_task.py gate`: returns a deterministic gate outcome plus a local Readiness Brief. +- `scripts/resolve_simple_task.py gate`: resolves normalized task facts into a + gate outcome, Readiness Brief, and pre-execution gate requirements. - `scripts/resolve_simple_task.py claim`: returns evidence requirements for strong status claims. - `scripts/suggest_support_skills.py`: returns generic method hints when the next move is still noisy. +- Script output contract: `text` for short operator summaries (default), `json` for nested or machine-consumed output, `tsv`/`csv` only for large flat tables; data on stdout, diagnostics on stderr; keep output bounded. ## Validation @@ -145,11 +179,11 @@ Keep the implementation posture simple: - Concrete bounded work completes in the same run unless `stop-with-reason` is explicit. - Stop output explains the exact violated condition and required evidence. - `trivial-skip` names the validation path directly or states the exact validation gap. -- Non-trivial work completes Initial Idea Ordering before `grill-me`. +- Non-trivial work completes Initial Idea Ordering before `/grill-me`. - The critical challenge runs before non-trivial action. - Non-trivial work has a Gate Evidence Ledger entry for each required row, or an explicit blocker. - Skipped gates record the reason that made the skip valid. - Blocked gates prevent completion, readiness, passing, fixed, or no-gap claims. -- Executable or evaluable behavior changes load `internal-tdd` before implementation when a meaningful seam exists. -- `addyosmani-code-simplification` loads only for an explicit simplification request or already-approved remediation after a passing behavior baseline exists. +- Executable or evaluable behavior changes load `/internal-tdd` before implementation when a meaningful seam exists. +- `/addyosmani-code-simplification` loads only for an explicit simplification request or already-approved remediation after a passing behavior baseline exists. - Positive claims rely on fresh evidence before completion. diff --git a/.github/skills/internal-gateway-simple-task/agents/openai.yaml b/.github/skills/internal-gateway-simple-task/agents/openai.yaml index 7fcd52f7..6b8052d2 100644 --- a/.github/skills/internal-gateway-simple-task/agents/openai.yaml +++ b/.github/skills/internal-gateway-simple-task/agents/openai.yaml @@ -1,4 +1,4 @@ interface: display_name: "Internal Gateway Simple Task" short_description: "Fast path for concrete repository tasks" - default_prompt: "Use $internal-gateway-simple-task for a concrete coding or non-coding task that should finish end-to-end in one bounded run. Start with bounded evidence, local idea ordering, and one compact clarification only when necessary. If the task changes executable or evaluable behavior, load `internal-tdd` before implementation. Use `grill-me` when the task is non-trivial, keep a compact gate evidence ledger for non-trivial work, run `internal-gateway-critical-master` before non-trivial action, stop with reason when the work becomes too large, costly, ambiguous, unsafe, or not locally verifiable, and use `superpowers-verification-before-completion` before positive claims." + default_prompt: "Use /internal-gateway-simple-task for a concrete coding or non-coding task that should finish end-to-end in one bounded run. Start with bounded evidence, local idea ordering, and one compact clarification only when necessary. If the task changes executable or evaluable behavior, load `/internal-tdd` before implementation. Use `/grill-me` only when a missing bounded fact blocks the active lane. Treat ordinary full-gate readiness as internal; stop only when an approval boundary or another named stop condition is present. Keep gate requirements distinct from the actual internal Gate Evidence Ledger. Run `/internal-gateway-critical-master` before non-trivial action. Stop with reason when the work becomes too large, costly, ambiguous, unsafe, or not locally verifiable. Use `/superpowers-verification-before-completion` before positive claims. Emit no more than four content lines and surface risks, blockers, gaps, or user actions only when present. Use --format json for complete diagnostic evidence." diff --git a/.github/skills/internal-gateway-simple-task/references/clarification-gate.md b/.github/skills/internal-gateway-simple-task/references/clarification-gate.md index e5e5f958..84f09cc2 100644 --- a/.github/skills/internal-gateway-simple-task/references/clarification-gate.md +++ b/.github/skills/internal-gateway-simple-task/references/clarification-gate.md @@ -4,7 +4,7 @@ Use this reference when the simple-task bundle must decide whether one focused c ## Core Rule -Default to the full gate for non-trivial work: bounded evidence, Initial Idea Ordering, one compact `grill-me` block when needed, critical challenge, then the local Readiness Brief. +Default to the full gate for non-trivial work: bounded evidence, Initial Idea Ordering, one compact `/grill-me` block when needed, critical challenge, then the internal readiness record. Complete the internal readiness record, then render only the compact user-facing projection. Do not paste the internal record or Gate Evidence. Skip the gate only with a Trivial-skip proof showing: @@ -31,7 +31,7 @@ If `1` or `2` is yes and `3-5` are no, use the full gate. If any of `3-5` is yes ## Single Clarification Limit -Ask at most one compact `grill-me` block for: +Ask at most one compact `/grill-me` block for: - missing file, path, or artifact target - missing input data or reproduction step @@ -40,16 +40,16 @@ Ask at most one compact `grill-me` block for: If that answer creates another dependent question set, stop with reason. -## `grill-me` Boundary +## `/grill-me` Boundary -`grill-me` may recover: +`/grill-me` may recover: - a missing path or artifact - a missing reproduction step - one bounded blocker - one missing local fact needed to execute -`grill-me` must not decide: +`/grill-me` must not decide: - staged workflow changes - architecture or design tradeoffs @@ -59,4 +59,4 @@ If that answer creates another dependent question set, stop with reason. ## Stop Conditions -Stop conditions are owned by `references/support-routing.md`; this gate adds only the per-question stop rules listed in Stop rules above. +Stop conditions are owned by `references/support-routing.md`; this gate adds only the per-question stop rules listed in Stop Conditions above. diff --git a/.github/skills/internal-gateway-simple-task/references/plan-mode.md b/.github/skills/internal-gateway-simple-task/references/plan-mode.md index 1c5813e3..23a1e919 100644 --- a/.github/skills/internal-gateway-simple-task/references/plan-mode.md +++ b/.github/skills/internal-gateway-simple-task/references/plan-mode.md @@ -18,6 +18,10 @@ Recommend a plan when one or more of these signals is present: - the task risks context pressure before validation can complete - the work centers on large exports, tables, logs, or broad mechanical change +The step, file-family, and validator counts are warning signals. Stop only +when their combination makes same-run completion unsafe or uneconomical; +do not classify from one count alone. + ## Procedure 1. Classify the cost or complexity signal. diff --git a/.github/skills/internal-gateway-simple-task/references/simple-lanes.md b/.github/skills/internal-gateway-simple-task/references/simple-lanes.md index aa871825..d02f926b 100644 --- a/.github/skills/internal-gateway-simple-task/references/simple-lanes.md +++ b/.github/skills/internal-gateway-simple-task/references/simple-lanes.md @@ -15,13 +15,17 @@ If cost or complexity exceeds same-run execution, stop and recommend a retained ## Output Shapes +Normal chat uses the compact user-facing projection from `SKILL.md`. Lane +evidence and the actual Gate Evidence Ledger remain internal unless the user +requests diagnostic JSON. + For `answer`, return the answer, evidence, uncertainty, and any validation gap when applicable. -For `edit`, return `lane`, `files-touched`, `gate-ledger`, validation, and residual risk. +For `edit`, return `lane`, `files-touched`, validation, and residual risk. -For `diagnose`, return `lane`, reproduced failure, root cause, fix or blocker, `gate-ledger`, and evidence. +For `diagnose`, return `lane`, reproduced failure, root cause, fix or blocker, and evidence. -For `validate`, return `lane`, check, result, `gate-ledger`, and any follow-up gap. +For `validate`, return `lane`, check, result, and any follow-up gap. For `stop-with-reason`, return: diff --git a/.github/skills/internal-gateway-simple-task/references/support-routing.md b/.github/skills/internal-gateway-simple-task/references/support-routing.md index c220cc1a..68375dbf 100644 --- a/.github/skills/internal-gateway-simple-task/references/support-routing.md +++ b/.github/skills/internal-gateway-simple-task/references/support-routing.md @@ -24,7 +24,7 @@ If no strong signal exists, stay local or stop with reason. | --- | --- | --- | | Missing intent, target path, input data, local context, or one blocker prevents starting | Ask one compact clarification block. | Stop if the answer would change scope, validation, cost, or risk. | | Bug, failing test, failing build, drift, or unexpected output | Reproduce first, then debug by falsifiable hypothesis. | Do not patch from correlation alone. | -| Executable behavior change | Load `internal-tdd` to classify `mandatory`, `recommended`, or `not suitable` routing when a meaningful executable or evaluable seam exists. | Do not force it onto pure prose or governance wording with no executable seam. | +| Executable behavior change | Load `/internal-tdd` to classify `mandatory`, `recommended`, or `not suitable` routing when a meaningful executable or evaluable seam exists. | Do not force it onto pure prose or governance wording with no executable seam. | | Existing diff needs findings or merge-readiness | Stop with reason because the work is no longer simple execution. | Do not turn simple validation into review. | | Orientation or unfamiliar code mapping | Stay descriptive and bounded. | Do not turn orientation into findings without concrete evidence. | | Architecture, workflow, cross-cutting impact, or blind spots dominate | Stop with reason. | Do not keep editing while the boundary is unsettled. | @@ -47,7 +47,7 @@ If the evidence gate would require broader staged work, stop with reason instead ## Advisory Helpers -Run `scripts/resolve_simple_task.py gate` when the task facts are already normalized and only the gate outcome plus Readiness Brief is noisy. +Run `scripts/resolve_simple_task.py gate` for compact operator text. Add `--format json` only when a tool or diagnostic flow needs the complete internal readiness record and gate requirements. Run `scripts/resolve_simple_task.py claim` before strong status claims when the evidence requirements are the noisy part. diff --git a/.github/skills/internal-gateway-simple-task/scripts/resolve_simple_task.py b/.github/skills/internal-gateway-simple-task/scripts/resolve_simple_task.py index 08eaa299..4347b7e8 100644 --- a/.github/skills/internal-gateway-simple-task/scripts/resolve_simple_task.py +++ b/.github/skills/internal-gateway-simple-task/scripts/resolve_simple_task.py @@ -23,6 +23,16 @@ "cross-file", "user-visible", ) + +STOP_REASON_TEXT = { + "plan-recommended": "The work needs a retained plan before execution.", + "review-shaped": "The request is findings-first rather than simple execution.", + "owner-ambiguous": "The controlling owner is not established.", + "clarification-overflow": "More than one dependent clarification is required.", + "approval-required": "Execution depends on a separate approval.", + "validation-gap": "The non-trivial task has no usable local validation path.", + "validation-path-missing": "The validation path or exact allowed gap is not named.", +} CLAIMS = ( "completion", "covered", @@ -162,6 +172,7 @@ def parse_args() -> argparse.Namespace: gate_parser.add_argument("--validation-obvious", action="store_true") gate_parser.add_argument("--validation-path", default="") gate_parser.add_argument("--validation-gap", default="") + gate_parser.add_argument("--approval-required", action="store_true") gate_parser.add_argument("--format", choices=("text", "json"), default="text") claim_parser = subparsers.add_parser( @@ -198,7 +209,7 @@ def infer_lane(lane: str, trivial_kind: str | None) -> str: return mapping.get(trivial_kind or "", "unspecified") -def build_gate_evidence( +def build_gate_requirements( *, gate_outcome: str, validation_path: str, @@ -281,50 +292,65 @@ def build_gate_decision( validation_obvious: bool, validation_path: str, validation_gap: str, + approval_required: bool = False, ) -> dict[str, object]: resolved_depth_keywords = detect_depth_keywords(prompt, depth_keywords) resolved_risks = sorted(set(risks)) resolved_lane = infer_lane(lane, trivial_kind) - stop_reasons: list[str] = [] - stop_for_material_boundary = bool(validation_gap) and any( - risk in {"architecture", "cross-file", "governance", "rollout"} - for risk in resolved_risks + + has_named_validation = bool(validation_path) + has_explicit_gap = bool(validation_gap) + is_trivial_candidate = trivial_kind in TRIVIAL_KINDS + missing_validation_evidence = ( + not has_named_validation and not has_explicit_gap ) + nontrivial_validation_gap = has_explicit_gap and not is_trivial_candidate + approval_boundary = approval_required + + reason_codes: list[str] = [] + blocking_reasons: list[str] = [] if needs_plan: - stop_reasons.append("plan-recommended") + reason_codes.append("plan-recommended") + blocking_reasons.append("plan-recommended") if needs_review: - stop_reasons.append("review-shaped") + reason_codes.append("review-shaped") + blocking_reasons.append("review-shaped") if needs_critical: - stop_reasons.append("critical-challenge-needed") + reason_codes.append("critical-challenge-needed") if owner_ambiguous: - stop_reasons.append("owner-ambiguous") + reason_codes.append("owner-ambiguous") + blocking_reasons.append("owner-ambiguous") if clarification_overflow: - stop_reasons.append("clarification-overflow") + reason_codes.append("clarification-overflow") + blocking_reasons.append("clarification-overflow") + if approval_boundary: + reason_codes.append("approval-required") + blocking_reasons.append("approval-required") for keyword in resolved_depth_keywords: - stop_reasons.append(f"depth-keyword:{keyword}") + reason_codes.append(f"depth-keyword:{keyword}") for risk in resolved_risks: - stop_reasons.append(f"material-risk:{risk}") - if stop_for_material_boundary: - stop_reasons.append("material-boundary-break") - - if ( - needs_plan - or needs_review - or owner_ambiguous - or clarification_overflow - or stop_for_material_boundary - ): + reason_codes.append(f"material-risk:{risk}") + if nontrivial_validation_gap: + reason_codes.append("validation-gap") + blocking_reasons.append("validation-gap") + if missing_validation_evidence: + reason_codes.append("validation-path-missing") + blocking_reasons.append("validation-path-missing") + if validation_obvious and missing_validation_evidence: + reason_codes.append("validation-path-missing") + + if blocking_reasons: gate_outcome = "stop-with-reason" elif ( - trivial_kind in TRIVIAL_KINDS + is_trivial_candidate and not resolved_depth_keywords and not resolved_risks and not needs_critical - and (validation_obvious or bool(validation_path) or bool(validation_gap)) + and (has_named_validation or has_explicit_gap) ): gate_outcome = "trivial-skip" - stop_reasons.append(f"trivial-kind:{trivial_kind}") + reason_codes.append(f"trivial-kind:{trivial_kind}") else: gate_outcome = "full-gate" @@ -333,7 +359,7 @@ def build_gate_decision( elif validation_gap: focused_validation_path = f"Validation gap: {validation_gap}" elif validation_obvious: - focused_validation_path = "Validation is obvious but not yet named." + focused_validation_path = "Validation path not yet named." else: focused_validation_path = "Validation path not yet identified." @@ -353,6 +379,16 @@ def build_gate_decision( else: approval = "explicit user approval before non-trivial operational work" + why_stopped = ( + STOP_REASON_TEXT[blocking_reasons[0]] if blocking_reasons else "" + ) + violated_condition = blocking_reasons[0] if blocking_reasons else "" + evidence_required = ( + "Provide the missing validation path, approval, owner, or bounded plan evidence." + if blocking_reasons + else "" + ) + readiness_brief = { "task": task, "goal": "Complete the current bounded task in one run when safe.", @@ -365,29 +401,32 @@ def build_gate_decision( "main_risk": main_risk, "stop_conditions": "Stop for complexity, cost, ambiguity, safety, approval, or validation gaps.", "approval": approval, + "why_stopped": why_stopped, + "violated_condition": violated_condition, + "evidence_required": evidence_required, } if gate_outcome == "trivial-skip": - gate_evidence = { + gate_requirements = { "validation": focused_validation_path, - "final_evidence": "Fresh evidence supporting the trivial claim.", + "final_evidence": "Fresh evidence supporting the final claim.", } else: - gate_evidence = build_gate_evidence( + gate_requirements = build_gate_requirements( gate_outcome=gate_outcome, validation_path=validation_path, validation_gap=validation_gap, clarification_required=needs_clarification, - stop_reasons=stop_reasons, + stop_reasons=reason_codes, ) return { "gate_outcome": gate_outcome, - "next_action": "stop" if gate_outcome == "stop-with-reason" else "execute", + "next_action": "stop" if blocking_reasons else "execute", "lane": resolved_lane, - "reason_codes": stop_reasons, - "needs_explicit_approval": gate_outcome != "trivial-skip", + "reason_codes": reason_codes, + "needs_explicit_approval": approval_boundary, "readiness_brief": readiness_brief, - "gate_evidence": gate_evidence, + "gate_requirements": gate_requirements, } @@ -406,36 +445,18 @@ def resolve_claim_requirements(claims: list[str]) -> list[dict[str, str]]: def render_gate_text(decision: dict[str, object]) -> None: brief = decision["readiness_brief"] - print(f"Gate outcome: {decision['gate_outcome']}") - print(f"Next action: {decision['next_action']}") - print(f"Lane: {decision['lane']}") - if decision["reason_codes"]: - print("Reason codes:") - for reason in decision["reason_codes"]: - print(f"- {reason}") - print("Readiness Brief:") - print(f"- Task: {brief['task']}") - print(f"- Goal: {brief['goal']}") - print(f"- Scope: {brief['scope']}") - print(f"- Anti-scope: {brief['anti_scope']}") - print(f"- Files expected: {brief['files_expected']}") - print(f"- Approach: {brief['approach']}") - print(f"- Executable behavior: {brief['executable_behavior']}") - print(f"- Validation path: {brief['validation_path']}") - print(f"- Main risk: {brief['main_risk']}") - print(f"- Stop conditions: {brief['stop_conditions']}") - print(f"- Approval: {brief['approval']}") - print("Gate Evidence:") - gate_evidence = decision["gate_evidence"] - if isinstance(gate_evidence, dict): - for key, value in gate_evidence.items(): - print(f"- {key}: {value}") - else: - for evidence in gate_evidence: - print( - f"- {evidence['gate']}: required={str(evidence['required']).lower()}; " - f"expected={evidence['expected_evidence']}" - ) + if decision["next_action"] == "stop": + print(f"🧭 Stop: {brief['why_stopped']}") + print(f"🎯 Boundary: {brief['violated_condition']}") + print(f"🧪 Needed: {brief['evidence_required']}") + print("✈️ Action: resolve the named boundary before execution.") + return + + print(f"🧭 {decision['gate_outcome']}: {brief['task']}") + print(f"🛠️ Scope: {brief['scope']}") + print(f"🧪 Check: {brief['validation_path']}") + if brief["main_risk"]: + print(f"⚠️ Risk: {brief['main_risk']}") def render_claim_text(claims: list[str], requirements: list[dict[str, str]]) -> None: @@ -463,6 +484,7 @@ def main() -> int: validation_obvious=args.validation_obvious, validation_path=args.validation_path, validation_gap=args.validation_gap, + approval_required=args.approval_required, ) if args.format == "json": print(json.dumps(decision, indent=2)) diff --git a/.github/skills/internal-gateway-writing-plans/SKILL.md b/.github/skills/internal-gateway-writing-plans/SKILL.md index cf9c1d84..826366d1 100644 --- a/.github/skills/internal-gateway-writing-plans/SKILL.md +++ b/.github/skills/internal-gateway-writing-plans/SKILL.md @@ -1,71 +1,66 @@ --- name: internal-gateway-writing-plans -description: Use when repository-owned work needs a short preflight before delegating retained writing to superpowers-writing-plans. +description: Use when repository-owned work needs approved implementation plan writing from an approved design or reviewed retained spec. --- # Internal Gateway Writing Plans ## Referenced skills -- `superpowers-writing-plans`: required owner after the repository preflight. +- `/superpowers-writing-plans`: required owner after the repository preflight. -Thin repository wrapper for retained writing. This skill records the local -handoff facts, delegates artifact decisions to `superpowers-writing-plans`, and -stops after the delegated outcome. +Thin repository wrapper for approved implementation-plan writing. This skill records the local handoff facts, delegates artifact decisions to `/superpowers-writing-plans`, treats the delegated result as a draft until the local acceptance gate passes, and stops after reporting the accepted plan path. ## When to use -- Use after the user approves retained spec or implementation-plan writing. +- Use after the user approves implementation-plan writing from an approved design or reviewed retained spec. ## When not to use -- Do not use for quick same-chat tasks, substantive ideation, execution, or - imported `superpowers-*` edits. +- Retained-spec writing stays in the brainstorming lane. +- Route same-chat work, ideation, plan review, execution, and imported `superpowers-*` maintenance to their existing owners. ## Contract -1. Capture the preflight: `Target`, `Anti-scope`, `Nearest owner`, - `Validation path`, `Stop conditions`, and `Observable acceptance`. -2. Load `superpowers-writing-plans` and let it create a plan, ask a blocking - clarification, redirect, or stop with a reason. Pass an explicit anti-scope - and the relevant owners already identified in the preflight so the delegated - plan avoids duplicate or speculative tasks at the source. Pass an explicit - delivery rule: the delegated plan must not tell the user to run `git add`, - `git commit`, or `git push`, and must not present committing changes as the - default next step unless the user explicitly asks for commit help. If the - delegated writing outcome persists a retained artifact, require timestamped local - naming with four-digit 24-hour time: plans use - `tmp/superpowers/plans/YYYY-MM-DD-HHMM-.md` and specs use - `tmp/superpowers/specs/YYYY-MM-DD-HHMM--design.md` such as - `2026-07-03-1143-`. -3. If a retained plan is created, verify execution-readiness and apply Plan - Authoring Discipline: ordered tasks, concrete file targets, clear edit - intent, validation commands or explicit gaps, stop conditions, and handoff - readiness. Reject the draft if any task duplicates an existing owner, adds - speculative scope, includes direct commit instructions without explicit - user approval, or lacks validation commands or an explicit validation gap. -4. Stop after the writing outcome and wait for the user's next choice. - -Preserve handoff quality with targeted rereads only when the delegation has a -real evidence gap. - -## Plan Authoring Discipline - -- Owner-first: before the delegated plan adds a task, confirm no existing - owner, skill, or validator already covers that responsibility; prefer a - reference over a duplicate. -- Single responsibility: each task must carry one clear deliverable tied to - the approved target; split or merge tasks that don't. -- Fail-fast and redirect: if the delegated plan adds speculative scope, - duplicates an existing owner, includes direct `git add`/`git commit`/`git push` - instructions without explicit user approval, or lacks validation commands or - an explicit validation gap, send it back for revision instead of accepting it. - -DRY, YAGNI, and TDD stay owned by `superpowers-writing-plans`; this section -adds only the owner-awareness and redirect gate that the delegated skill does -not enforce. +### Preflight Gate + +Capture the preflight: `Target`, `Anti-scope`, `Nearest owner`, +`Validation path`, `Stop conditions`, and `Observable acceptance`. + +Completion criterion: all six preflight facts are present and no fact is missing or explicitly recorded as a gap. + +### Delegated Draft Gate + +Load `/superpowers-writing-plans` and let it create a plan, ask a blocking clarification, redirect, or stop with a reason. Pass an explicit anti-scope and the relevant owners already identified in the preflight so the delegated plan avoids duplicate or speculative tasks at the source. Pass an explicit delivery rule: the delegated plan must not contain `git add`, `git commit`, or `git push` steps or instructions, and must not present committing changes as the default next step unless the user explicitly asks for commit help. The delegated writing outcome persists as a draft-only artifact under `tmp/superpowers/plans/YYYY-MM-DD-HHMM-.md`. + +Completion criterion: one delegated plan artifact exists under the timestamped plan path and is marked draft-only. + +### Local Acceptance Gate + +Delegated output remains draft-only until objective checks pass and human judgment checks pass. If either fails, revise the draft in place. + +Objective checks: run `python3 scripts/validate_plan.py `. A non-zero finding result keeps the artifact draft-only. Follow mechanical validation with human checks. + +Human judgment checks: owner duplication, speculative scope, coherent task boundaries, one responsibility per task, edit intent, focused validation, stop conditions, and handoff readiness. Reject unapproved simplification or duplicated execution workflow. + +Completion criterion: objective checks pass and human judgment checks pass. + +### Writing Stop + +After the local acceptance gate passes, name `/internal-gateway-execute-plans` as the next owner. Stop after reporting the accepted plan path and wait for the user's next choice. + +Completion criterion: accepted plan path is reported and `/internal-gateway-execute-plans` is named as the next owner. + +Preserve handoff quality with targeted rereads only when the delegation has a real evidence gap. + +## No-Commit Rule + +- The skill must never run `git add`, `git commit`, `git push`, or any other git mutation while creating, persisting, or handing off plans. Retained artifacts stay uncommitted under `tmp/superpowers/`; the user reviews and commits them personally. +- This rule is mandatory. The user may bypass it only with an explicit request for commit help in the current task; state the bypass in the outcome summary. +- The produced plan contains no Git mutation steps or default commit advice. ## Validation - Confirm the delegated plan carries ordered tasks, concrete file targets, clear edit intent, validation commands or explicit gaps, no duplicate-owner or speculative-scope drift, and no direct commit instructions unless the user explicitly asked for commit help. +- Confirm no git mutation ran while producing the writing outcome and that retained artifacts remain uncommitted, unless the user explicitly asked for commit help. - `git diff --check` diff --git a/.github/skills/internal-gateway-writing-plans/agents/openai.yaml b/.github/skills/internal-gateway-writing-plans/agents/openai.yaml index c8ffd90b..e53bb2c3 100644 --- a/.github/skills/internal-gateway-writing-plans/agents/openai.yaml +++ b/.github/skills/internal-gateway-writing-plans/agents/openai.yaml @@ -1,15 +1,12 @@ interface: display_name: "Internal Gateway Writing Plans" - short_description: "Repository preflight wrapper for superpowers-writing-plans" + short_description: "Approved implementation-plan writing" default_prompt: >- - Use $internal-gateway-writing-plans to capture Target, Anti-scope, + Use /internal-gateway-writing-plans to capture Target, Anti-scope, Nearest owner, Validation path, Stop conditions, and Observable acceptance. - Then load $superpowers-writing-plans and delegate the writing outcome. If a - retained artifact is persisted, require timestamped local naming with - `tmp/superpowers/plans/YYYY-MM-DD-HHMM-.md` for plans and - `tmp/superpowers/specs/YYYY-MM-DD-HHMM--design.md` for specs, for - example `2026-07-03-1143-`. - If a retained plan is created, check ordered tasks, concrete file targets, edit - intent, validation commands or explicit gaps, duplicate-owner or - speculative-scope drift, stop conditions, and handoff readiness. Stop after - the writing outcome. + Then load /superpowers-writing-plans and delegate the approved implementation plan + writing outcome. The delegated draft remains draft-only until the local acceptance gate + passes. Persist the plan under `tmp/superpowers/plans/YYYY-MM-DD-HHMM-.md`. + After the local acceptance gate passes, name /internal-gateway-execute-plans as the + next owner. Stop after reporting the accepted plan path. Observe the No-Commit Rule: + no git mutations during writing. diff --git a/.github/skills/internal-gateway-writing-plans/fixtures/2026-07-25-1829-invalid-plan.md b/.github/skills/internal-gateway-writing-plans/fixtures/2026-07-25-1829-invalid-plan.md new file mode 100644 index 00000000..cccb9af8 --- /dev/null +++ b/.github/skills/internal-gateway-writing-plans/fixtures/2026-07-25-1829-invalid-plan.md @@ -0,0 +1,31 @@ +# Invalid Implementation Plan + +## Missing Required Fields + +This plan has no required preflight information. + +### Task 2: Wrong order + +**Files:** + +- No concrete targets here + +Steps: + +1. Do something without validation + +### Task 1: Also wrong order + +**Files:** + +- Still no targets + +Steps: + +1. Do more things +2. git add . +3. git commit -m "changes" + +## Handoff + +Route to superpowers-executing-plans after acceptance. diff --git a/.github/skills/internal-gateway-writing-plans/fixtures/2026-07-25-1829-valid-plan.md b/.github/skills/internal-gateway-writing-plans/fixtures/2026-07-25-1829-valid-plan.md new file mode 100644 index 00000000..9d4849fc --- /dev/null +++ b/.github/skills/internal-gateway-writing-plans/fixtures/2026-07-25-1829-valid-plan.md @@ -0,0 +1,50 @@ +# Valid Implementation Plan + +## Preflight + +- Target: example feature +- Anti-scope: no unrelated cleanup +- Nearest owner: .github/skills/example/SKILL.md +- Validation path: focused pytest +- Stop conditions: none +- Observable acceptance: tests pass + +### Task 1: Add example test + +**Files:** + +- Modify: tests/example/test_example.py + +Steps: + +1. Add a failing test +2. Run pytest +3. Implement the feature +4. Run pytest again + +Validation: + +```bash +rtk pytest -q tests/example/test_example.py +``` + +### Task 2: Implement example feature + +**Files:** + +- Modify: .github/skills/example/SKILL.md + +Steps: + +1. Update the skill description +2. Run validation + +Validation: + +```bash +rtk pytest -q tests/example/test_example.py +``` + +## Handoff + +Route to `internal-gateway-execute-plans` after acceptance. diff --git a/.github/skills/internal-gateway-writing-plans/scripts/validate_plan.py b/.github/skills/internal-gateway-writing-plans/scripts/validate_plan.py new file mode 100644 index 00000000..356eabcc --- /dev/null +++ b/.github/skills/internal-gateway-writing-plans/scripts/validate_plan.py @@ -0,0 +1,155 @@ +#!/usr/bin/env python3 +"""Deterministic objective validation for retained implementation Plans.""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + + +FILENAME_PATTERN = re.compile(r"^\d{4}-\d{2}-\d{2}-\d{4}-[a-z0-9-]+\.md$") +TASK_HEADING_PATTERN = re.compile(r"^###\s+Task\s+(\d+):") +FILES_BLOCK_PATTERN = re.compile( + r"\*\*Files:\*\*\s*\n((?:\s*-\s+.+\n?)+)", re.MULTILINE +) +FILE_TARGET_PATTERN = re.compile(r"\.github/|tests/|AGENTS\.md|Makefile") +VALIDATION_PATTERN = re.compile(r"rtk\s+\S+|python3\s+\S+|pytest|make\s+\S+") +GIT_MUTATION_PHRASES = ( + "git add", + "git commit", + "git push", + "git merge", +) + + +def _check_filename(path: Path) -> str | None: + if not FILENAME_PATTERN.match(path.name): + return "filename" + return None + + +def _check_preflight(text: str) -> str | None: + required = ("Target", "Anti-scope", "Validation path", "Stop conditions", "Observable acceptance") + for marker in required: + if marker not in text: + return "preflight" + return None + + +def _extract_tasks(text: str) -> list[tuple[int, str]]: + tasks: list[tuple[int, str]] = [] + lines = text.splitlines() + for idx, line in enumerate(lines): + match = TASK_HEADING_PATTERN.match(line) + if match: + task_num = int(match.group(1)) + task_body = "\n".join(lines[idx:]) + next_match = TASK_HEADING_PATTERN.search(task_body[len(line) + 1:]) + if next_match: + task_body = "\n".join(lines[idx:idx + 1 + next_match.start()]) + tasks.append((task_num, task_body)) + return tasks + + +def _check_ordered_tasks(tasks: list[tuple[int, str]]) -> str | None: + if not tasks: + return "ordered_tasks" + nums = [t[0] for t in tasks] + if nums != sorted(nums) or len(nums) != len(set(nums)): + return "ordered_tasks" + return None + + +def _check_file_targets(tasks: list[tuple[int, str]]) -> str | None: + for _, body in tasks: + files_match = FILES_BLOCK_PATTERN.search(body) + if not files_match: + return "file_targets" + files_block = files_match.group(1) + if not FILE_TARGET_PATTERN.search(files_block): + return "file_targets" + return None + + +def _check_validation(tasks: list[tuple[int, str]]) -> str | None: + for _, body in tasks: + if not VALIDATION_PATTERN.search(body) and "validation gap" not in body.lower(): + return "validation" + return None + + +def _check_git_mutation(text: str) -> str | None: + lower = text.lower() + for phrase in GIT_MUTATION_PHRASES: + if phrase in lower: + return "git_mutation" + return None + + +def _check_execution_owner(text: str) -> str | None: + if "internal-gateway-execute-plans" not in text: + return "execution_owner" + if "superpowers-executing-plans" in text and "internal-gateway-execute-plans" not in text: + return "execution_owner" + return None + + +def validate_plan(path: Path) -> list[dict[str, str]]: + findings: list[dict[str, str]] = [] + text = path.read_text(encoding="utf-8") + + filename_code = _check_filename(path) + if filename_code: + findings.append({"code": filename_code, "message": f"invalid basename: {path.name}"}) + + preflight_code = _check_preflight(text) + if preflight_code: + findings.append({"code": preflight_code, "message": "missing preflight fields"}) + + tasks = _extract_tasks(text) + ordered_code = _check_ordered_tasks(tasks) + if ordered_code: + findings.append({"code": ordered_code, "message": "tasks not numbered and strictly increasing"}) + + file_code = _check_file_targets(tasks) + if file_code: + findings.append({"code": file_code, "message": "task missing concrete file targets"}) + + validation_code = _check_validation(tasks) + if validation_code: + findings.append({"code": validation_code, "message": "task missing validation command or gap"}) + + git_code = _check_git_mutation(text) + if git_code: + findings.append({"code": git_code, "message": "plan contains git mutation instruction"}) + + owner_code = _check_execution_owner(text) + if owner_code: + findings.append({"code": owner_code, "message": "missing or incorrect execution owner handoff"}) + + return findings + + +def main() -> int: + if len(sys.argv) != 2: + print("Usage: validate_plan.py ", file=sys.stderr) + return 2 + + path = Path(sys.argv[1]) + if not path.exists(): + print(f"Error: {path} does not exist", file=sys.stderr) + return 2 + + findings = validate_plan(path) + if findings: + for finding in findings: + print(f"{finding['code']}: {finding['message']}") + return 1 + + print("PASS") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/skills/internal-gcp-governance/SKILL.md b/.github/skills/internal-gcp-governance/SKILL.md index 965c6a8c..00aafaa3 100644 --- a/.github/skills/internal-gcp-governance/SKILL.md +++ b/.github/skills/internal-gcp-governance/SKILL.md @@ -1,20 +1,16 @@ --- name: internal-gcp-governance -description: Use when the user needs Google Cloud governance guidance for IAM operating models, workload identity federation, service account boundaries, Org Policy design, inheritance strategy, security guardrails, exception handling, or other controls that define what principals can do after the GCP structure is chosen. +description: Use when the user needs Google Cloud governance guidance for IAM operating models, workload identity federation, service account boundaries, Org Policy design, inheritance strategy, security guardrails, or exception handling that define what principals can do after the GCP structure is chosen. Do not use for org, folder, project, or Shared VPC layout; monitoring, backup, or rollout validation; or materially ambiguous requests with no clear governance deliverable. --- # Internal GCP Governance -## Referenced skills - -- `internal-gcp-strategic`: route back when direction or tradeoff framing is still unsettled. -- `internal-gcp-organization-structure`: route when org, folder, project, Shared VPC, or topology layout is the main decision. -- `internal-gcp-operations`: route when monitoring, reporting, backup, inventory, or post-rollout validation is the main need. - Use this skill when the next need is to define or review Google Cloud identity, access, and guardrail decisions. This skill owns governance logic after the broad structure is known. It helps separate org or folder guardrails from project-level grants and keeps permission decisions auditable. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-gcp`. + ## When to use - The user needs IAM model guidance across folders or projects. @@ -22,13 +18,6 @@ This skill owns governance logic after the broad structure is known. It helps se - The user needs Org Policy, inheritance, or security-guardrail guidance. - The user needs a review of guardrail design, exceptions, or access governance. -## When not to use - -- The main problem is org, folder, project, or Shared VPC layout. -- The main problem is strategic option framing before the governance surface is clear. -- The main problem is monitoring, reporting, backup, inventory, or post-rollout validation. -- The task is implementation-only. - ## Main domains covered - IAM operating model @@ -70,15 +59,6 @@ For broader asks, return: - exception or blast-radius note - what should be validated before rollout -## Relationship to adjacent skills - -- `internal-gcp-strategic` - Use first when the user still needs option framing or lens selection. -- `internal-gcp-organization-structure` - Use when the governance question is actually about where a capability should live. -- `internal-gcp-operations` - Use when the next need is preflight, reporting, validation, inventory, or operational evidence after the governance design is chosen. - ## Common mistakes | Mistake | Why it matters | Instead | @@ -96,4 +76,4 @@ For broader asks, return: - Confirm the recommended mechanism is clear about whether it prevents, grants, or constrains workload identity. - Confirm service-account and human-access boundaries are explicit when access crosses project or folder boundaries. - Confirm staged rollout validation is named before high-blast-radius Org Policy or IAM changes. -- Confirm the answer says when operational proof should move to `internal-gcp-operations`. +- Confirm out-of-scope needs, such as operational proof or structure placement, are identified as outside this lane instead of being answered here. diff --git a/.github/skills/internal-gcp-operations/SKILL.md b/.github/skills/internal-gcp-operations/SKILL.md index 4a30cf1d..a01e2721 100644 --- a/.github/skills/internal-gcp-operations/SKILL.md +++ b/.github/skills/internal-gcp-operations/SKILL.md @@ -1,20 +1,16 @@ --- name: internal-gcp-operations -description: Use when the user needs Google Cloud operational guidance for monitoring, logging, backup and restore, DR validation, asset inventory, preflight checks, post-rollout validation, reporting, or audit evidence after a structure or governance decision has already been made. +description: Use when the user needs Google Cloud operational guidance for monitoring, logging, backup and restore, DR validation, asset inventory, preflight checks, post-rollout validation, reporting, or audit evidence after a structure or governance decision is made. Do not use for org, folder, project, or Shared VPC layout; IAM, workload identity, or Org Policy design; or materially ambiguous requests with no clear operational deliverable. --- # Internal GCP Operations -## Referenced skills - -- `internal-gcp-strategic`: route back when direction or tradeoff framing is still unsettled. -- `internal-gcp-organization-structure`: route when org, folder, project, Shared VPC, or topology layout is the main decision. -- `internal-gcp-governance`: route when IAM, workload identity, service account, or Org Policy design is the main decision. - Use this skill when the next need is to validate, observe, or operationalize a GCP platform decision. This skill owns the operational side of the platform: monitoring, evidence, inventory, preflight, and post-rollout verification. It does not replace strategic framing, structure design, or governance design. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-gcp`. + ## When to use - The user needs operational readiness guidance after a design choice. @@ -23,13 +19,6 @@ This skill owns the operational side of the platform: monitoring, evidence, inve - The user needs preflight or post-rollout validation patterns. - The user needs operational guidance for Cloud Run or Cloud Functions after the broader GCP structure and governance choices are already settled. -## When not to use - -- The main problem is still choosing the high-level direction. -- The main problem is org, folder, project, or Shared VPC structure. -- The main problem is IAM, workload identity, service account, or Org Policy design. -- The task is a narrow implementation change with no operational design question. - ## Main domains covered - monitoring and observability posture @@ -72,15 +61,6 @@ For broader asks, return: - recovery, DR, or inventory note when relevant - open operational risks -## Relationship to adjacent skills - -- `internal-gcp-strategic` - Use first when the core decision is still unsettled. -- `internal-gcp-organization-structure` - Use when the operations question is actually about org, project, or topology placement. -- `internal-gcp-governance` - Use when the operations question is actually about IAM, workload identity, Org Policy, or guardrail design rather than validation. - Until a dedicated GCP serverless owner is justified by real usage, keep Cloud Run and Cloud Functions operational readiness, validation, and evidence questions in this skill instead of inventing a fifth lane for symmetry alone. ## Common mistakes @@ -90,7 +70,7 @@ Until a dedicated GCP serverless owner is justified by real usage, keep Cloud Ru | Treating monitoring as proof that restore or recovery works | Healthy metrics do not prove recovery viability | Keep monitoring evidence, backup proof, and restore proof as separate lines | | Skipping preflight for high-blast-radius rollout | IAM, Org Policy, or Shared VPC regressions surface too late | Define rollout unit, preflight checks, rollback trigger, and owner before rollout | | Reporting only control intent without operational evidence | The platform appears compliant without proof that it works | Record what Monitoring, Logging, asset inventory, or recovery exercises actually showed | -| Mixing validation advice with new governance design instead of keeping the boundary clear | The operations skill stops being a reliable validation owner | Keep new Org Policy or IAM design in `internal-gcp-governance` and validate it here | +| Mixing validation advice with new governance design instead of keeping the boundary clear | The operations skill stops being a reliable validation owner | Keep new Org Policy or IAM design out of the validation answer and validate the chosen design here | | Giving a DR answer without making the business criticality assumption visible | Recovery guidance can be overbuilt or incomplete | State the assumed criticality, RTO, or RPO before recommending the evidence path | | Treating one successful rollout wave as proof for all folders or projects | Wider inheritance or network paths can still fail differently | Validate the first safe unit and widen only after recording real evidence | diff --git a/.github/skills/internal-gcp-organization-structure/SKILL.md b/.github/skills/internal-gcp-organization-structure/SKILL.md index 1d89a0c7..2bd91ff8 100644 --- a/.github/skills/internal-gcp-organization-structure/SKILL.md +++ b/.github/skills/internal-gcp-organization-structure/SKILL.md @@ -1,20 +1,16 @@ --- name: internal-gcp-organization-structure -description: Use when the user needs Google Cloud control-plane or platform-structure guidance for org or folder layout, billing-account model, project segmentation, Shared VPC host and service project topology, environment segmentation, platform-level network structure, or other layout decisions that shape how GCP is organized before implementation. +description: Use when the user needs Google Cloud control-plane or platform-structure guidance for org or folder layout, billing-account model, project segmentation, Shared VPC host and service project topology, environment segmentation, or platform-level network topology that shapes how GCP is organized before implementation. Do not use for IAM, service account, workload identity, or Org Policy design; monitoring, backup, or rollout validation; or materially ambiguous requests with no clear structural deliverable. --- # Internal GCP Organization Structure -## Referenced skills - -- `internal-gcp-strategic`: route back when direction or tradeoff framing is still unsettled. -- `internal-gcp-governance`: route IAM, service account, workload identity, or Org Policy design. -- `internal-gcp-operations`: route monitoring, backup, reporting, inventory, or post-rollout validation. - Use this skill when the next need is to design or review how GCP is structured at org and platform level. This skill owns GCP layout decisions, not generic strategy and not detailed IAM or monitoring implementation. It helps translate a platform goal into org, folder, project, Shared VPC, topology, and rollout structure. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-gcp`. + ## When to use - The user is shaping or reviewing GCP org or folder hierarchy. @@ -23,13 +19,6 @@ This skill owns GCP layout decisions, not generic strategy and not detailed IAM - The user needs platform-level network or regional structure guidance. - The user needs rollout-scope guidance for structural GCP change. -## When not to use - -- The question is mainly IAM, service account, workload identity, or Org Policy logic. -- The task is mainly monitoring, backup, reporting, or post-rollout validation. -- The user only needs generic strategic comparison with no concrete structure question. -- The task is already implementation-focused. - ## Main domains covered - org hierarchy @@ -74,25 +63,15 @@ For broader asks, return: - recommended placement model - smallest safe rollout unit - main risks -- what should move next to `internal-gcp-governance` or `internal-gcp-operations` - -## Relationship to adjacent skills - -- `internal-gcp-strategic` - Use first when the user still needs broader decision framing or lens selection. -- `internal-gcp-governance` - Use when the structural decision is accepted and the next need is IAM, workload identity federation, Org Policy, or guardrail definition. -- `internal-gcp-operations` - Use when the structure is accepted and the next need is validation, monitoring, backup, inventory, or operational evidence. ## Common mistakes | Mistake | Why it matters | Instead | | --- | --- | --- | | Proposing org or project layouts without a rollout scope | Structural changes are hard to unwind if staged poorly | Name the smallest safe rollout unit: folder, project set, or region set | -| Mixing Shared VPC placement and IAM design into one vague answer | Structure and governance review get blurred together | Keep host and service project placement here and move access design to `internal-gcp-governance` | +| Mixing Shared VPC placement and IAM design into one vague answer | Structure and governance review get blurred together | Keep host and service project placement here and keep access design out of the structure answer | | Treating billing layout as an afterthought when it changes ownership or blast radius | Finance and platform decisions drift together and become hard to review | Make billing ownership explicit when it differs from project or folder ownership | -| Using structure answers to sneak in Org Policy or IAM design without separating the concerns | The lane boundary becomes unreliable | State where the capability lives and hand off what controls or permissions apply | +| Using structure answers to sneak in Org Policy or IAM design without separating the concerns | The lane boundary becomes unreliable | State where the capability lives and keep what controls or permissions apply out of the structure answer | | Ignoring region or residency implications when they materially shape layout | Project or Shared VPC placement can violate real requirements | Make sovereignty, region choice, and continuity assumptions explicit | | Recommending central host projects without naming who operates them | Shared infrastructure becomes a vague platform bucket | State the host-project owner and which service projects depend on it | @@ -102,4 +81,4 @@ For broader asks, return: - Confirm the smallest safe rollout unit is named and matches the proposed structural change. - Confirm region or residency implications are explicit when they shape folder, project, or network placement. - Confirm billing ownership and platform ownership are separated when both appear in the recommendation. -- Confirm the next handoff is clear when the user now needs governance controls or operational proof. +- Confirm out-of-scope needs, such as governance controls or operational proof, are identified as outside this lane instead of being answered here. diff --git a/.github/skills/internal-gcp-strategic/agents/openai.yaml b/.github/skills/internal-gcp-strategic/agents/openai.yaml deleted file mode 100644 index 32288897..00000000 --- a/.github/skills/internal-gcp-strategic/agents/openai.yaml +++ /dev/null @@ -1,4 +0,0 @@ -interface: - display_name: "Internal GCP Strategic" - short_description: "GCP decision framing and tradeoff support" - default_prompt: "Use $internal-gcp-strategic to frame this GCP decision, keep the answer proportional, and activate only the lenses that matter." diff --git a/.github/skills/internal-gcp-strategic/SKILL.md b/.github/skills/internal-gcp/SKILL.md similarity index 57% rename from .github/skills/internal-gcp-strategic/SKILL.md rename to .github/skills/internal-gcp/SKILL.md index 0131477f..dbc746ba 100644 --- a/.github/skills/internal-gcp-strategic/SKILL.md +++ b/.github/skills/internal-gcp/SKILL.md @@ -1,29 +1,19 @@ --- -name: internal-gcp-strategic -description: Use when the user needs high-level Google Cloud platform decision support or tradeoff framing before implementation, and the next step is not yet structure, governance, operations, or code delivery. +name: internal-gcp +description: Use when a Google Cloud task cannot be routed confidently to a specific GCP skill because the request is materially ambiguous, has multiple GCP domains with no clear primary owner, or requires clarification before selecting the correct specialist, or when the user needs high-level Google Cloud platform decision support or tradeoff framing before implementation. Do not use for clearly scoped organization structure, governance or IAM, or operations or validation tasks. --- -# Internal GCP Strategic +# Internal GCP -## Referenced skills - -- `internal-gcp-organization-structure`: route when org, folder, project, Shared VPC, or topology layout becomes the decision. -- `internal-gcp-governance`: route when IAM, workload identity, service account, or Org Policy design becomes the decision. -- `internal-gcp-operations`: route when monitoring, backup, reporting, inventory, or post-rollout validation becomes the decision. -- `internal-terraform`: route implementation work in Terraform or OpenTofu. -- `internal-python-script`: route implementation work in standalone Python automation. -- `internal-bash-script`: route implementation work in standalone Bash automation. - -Use this skill when the main need is to reason about a GCP decision before implementation. - -This is a strategic support skill. It helps frame the decision, compare realistic options, expose tradeoffs, and recommend a direction. It does not implement the change and it does not choose Terraform, Python, or Bash on behalf of the user. +Fallback router for Google Cloud tasks that cannot be assigned confidently to one specialist, and strategic support skill for high-level GCP decision framing. Do not activate only because the task concerns Google Cloud; activate only when material routing uncertainty blocks owner selection or when the user needs decision support before the next step is structure, governance, operations, or delivery. ## When to use -- The user needs GCP decision support before execution. -- Multiple GCP approaches are credible and tradeoffs matter. -- The user wants a recommendation grounded in current Google Cloud guidance. -- The user wants high-level support for platform, organization, governance, resilience, cost, or operational decisions. +- Material ambiguity prevents selecting one primary GCP specialist. +- Multiple GCP domains are material and no primary owner can be identified safely. +- The user explicitly invokes `$internal-gcp`. +- The task asks which GCP lane should own the work before requesting a domain solution. +- The user needs high-level GCP decision support or tradeoff framing before implementation. ## When not to use @@ -32,15 +22,34 @@ This is a strategic support skill. It helps frame the decision, compare realisti - The task is purely post-rollout validation or evidence gathering. - The request is narrow and operational with no real decision to frame. -## Main domains covered +## Routing threshold + +Activate only when at least one holds: + +- the request is materially ambiguous and clarification is required before a GCP owner can be selected; +- multiple GCP domains are material and no primary owner can be identified safely; +- the task asks which GCP problem-solving lane should own the work; +- the user needs strategic decision framing and the next step is not yet structure, governance, operations, or delivery. + +Do not activate when one specialist clearly owns the next step; route directly to that specialist instead. Explicit `$internal-gcp` invocation remains valid. + +## Handoffs -- platform and control-plane decision framing -- Google Cloud Architecture Framework guidance at decision level -- org, folder, project, and billing-model implications at decision level -- governance and identity implications at decision level -- operational implications at decision level -- resilience and continuity implications at decision level -- cost-value and FinOps implications at decision level +| To | Owns | +|---|---| +| `internal-gcp-organization-structure` | org, folder, project, billing-account, Shared VPC, and platform topology layout | +| `internal-gcp-governance` | IAM, workload identity, service account, Org Policy, and guardrail design | +| `internal-gcp-operations` | monitoring, validation, backup, recovery, inventory, reporting, evidence | + +## Dispatch contract + +1. State the routing uncertainty. +2. Identify candidate GCP owners. +3. Select the minimum specialist set. +4. Keep strategic comparison here only while needed to choose the owner. +5. Hand the resolved task to the primary specialist instead of retaining ownership. + +Load `references/routing-matrix.md` for the routing decision tree. Load `references/lens-playbook.md` when the fallback trigger fires and the choice of GCP owner or lens needs structured comparison, or when the user wants a deeper decision-framing aid. ## Optional lens activation @@ -70,8 +79,6 @@ Rules: - If another lens would materially improve the recommendation, suggest it briefly instead of forcing it. - Keep the active lenses explicit when more than one is in play. -Load `references/lens-playbook.md` when the user wants a deeper framing aid or when the choice of lenses is not obvious. - ## Optional BC/DR lens BC/DR is optional. @@ -138,17 +145,6 @@ Include: - main risks and blast radius - validation or follow-up path -## Relationship to adjacent skills - -- `internal-gcp-organization-structure` - Use when the next need is org or folder layout, billing-account model, project topology, Shared VPC placement, or platform structure. -- `internal-gcp-governance` - Use when the next need is IAM model, workload identity federation, Org Policy design, or guardrail definition. -- `internal-gcp-operations` - Use when the next need is Cloud Monitoring, backup, DR validation, inventory, reporting, preflight, or post-rollout checks. -- `internal-terraform`, `internal-python-script`, `internal-bash-script` - Use when the decision is settled and implementation begins. - ## Common mistakes | Mistake | Why it matters | Instead | @@ -158,12 +154,14 @@ Include: | Recommending a direction without current-source verification when freshness matters | Product limits, Org Policy behavior, or regional support may have changed | Call out the freshness dependency and say which Google Cloud facts still need current verification | | Confusing decision support with implementation guidance | The user loses the strategic framing they asked for | Keep the answer at decision level and hand off only after the direction is chosen | | Expanding into tool or automation selection when the user did not ask for it | The response drifts from platform tradeoffs into delivery detail | Keep the recommendation centered on the GCP choice, not the tooling | -| Giving generic best-practice advice without context, tradeoff, or cost implication | Generic advice is hard to act on and easy to misapply | Tie the recommendation to assumptions, viable options, and cost-value consequences | +| Activating the fallback when one specialist clearly owns the next step | The router delays work a direct specialist should own | Route directly to the specialist and keep the fallback for genuine uncertainty | ## Validation +- State why the request could not be assigned to one primary GCP specialist, or name the decision being framed. +- Confirm the selected specialist set is the minimum needed to resolve the uncertainty. - Confirm the decision statement is explicit and narrow enough that the next owner is obvious. - Confirm assumptions, active lenses, and the main tradeoff are named instead of implied. - Confirm the recommendation includes reversibility or blast-radius guidance when the choice is hard to unwind. - Confirm cost-value or operational impact is called out when it materially changes the recommendation. -- Confirm the answer states when freshness matters and which current Google Cloud fact still needs validation. +- Confirm the answer states when freshness matters and which current Google Cloud fact still needs validation. \ No newline at end of file diff --git a/.github/skills/internal-gcp/agents/openai.yaml b/.github/skills/internal-gcp/agents/openai.yaml new file mode 100644 index 00000000..9d1c21a3 --- /dev/null +++ b/.github/skills/internal-gcp/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "Internal GCP" + short_description: "GCP routing and strategic decision support" + default_prompt: "Use $internal-gcp to route an unclear GCP task to the minimum specialist set, or to frame a Google Cloud decision when the next step is not yet structure, governance, operations, or delivery." \ No newline at end of file diff --git a/.github/skills/internal-gcp-strategic/references/lens-playbook.md b/.github/skills/internal-gcp/references/lens-playbook.md similarity index 100% rename from .github/skills/internal-gcp-strategic/references/lens-playbook.md rename to .github/skills/internal-gcp/references/lens-playbook.md diff --git a/.github/skills/internal-gcp/references/routing-matrix.md b/.github/skills/internal-gcp/references/routing-matrix.md new file mode 100644 index 00000000..7b5371a3 --- /dev/null +++ b/.github/skills/internal-gcp/references/routing-matrix.md @@ -0,0 +1,29 @@ +# GCP Routing Scenario Matrix + +## Fallback-positive cases + +| Scenario | Why no primary owner | +|---|---| +| Underspecified cross-org control problem mixing structure, governance, and operations with no clear primary deliverable | The request names multiple GCP domains but does not identify which deliverable takes priority. | +| GCP platform question asking which lane should own the work without naming structure, governance, or operations | The user has not selected a domain; the fallback must clarify the lane before any specialist can engage. | +| Broad Google Cloud adoption review where the user wants a general health assessment across all domains | No single specialist owns a cross-domain health review; the fallback selects the minimum set. | + +## Direct-specialist negative cases + +| Scenario | Direct owner | Reason | +|---|---|---| +| Org, folder, project, or Shared VPC layout | `internal-gcp-organization-structure` | The deliverable is a structural placement decision. | +| IAM, workload identity, service account, or Org Policy design | `internal-gcp-governance` | The deliverable is a guardrail or identity boundary. | +| Monitoring, backup, DR validation, inventory, or reporting | `internal-gcp-operations` | The deliverable is operational evidence or continuity proof. | + +## Multi-domain primary-owner cases + +| Scenario | Primary owner | Secondary | Reason | +|---|---|---|---| +| Project or Shared VPC design with later IAM work | `internal-gcp-organization-structure` | `internal-gcp-governance` | The first deliverable is placement; IAM and Org Policy follow once the structure is settled. | +| Org Policy or IAM rollout evidence | `internal-gcp-governance` | `internal-gcp-operations` | The first deliverable is governance design; operations validates the rollout. | +| Structural change needing continuity proof | `internal-gcp-organization-structure` | `internal-gcp-operations` | The first deliverable is layout; operations confirms the recovery posture after placement. | + +## Review rule + +Prefer a direct specialist whenever a reasonable reviewer can name one primary owner from the request itself. Activate the fallback only when the request does not identify a primary owner and clarification is required before a specialist can engage. The fallback is not a prerequisite for ordinary GCP work and must never activate all GCP skills by default. \ No newline at end of file diff --git a/.github/skills/internal-github-action-composite/SKILL.md b/.github/skills/internal-github-action-composite/SKILL.md index 591403e1..9ca21db9 100644 --- a/.github/skills/internal-github-action-composite/SKILL.md +++ b/.github/skills/internal-github-action-composite/SKILL.md @@ -1,6 +1,6 @@ --- name: internal-github-action-composite -description: Use when creating or modifying a reusable GitHub composite action under `.github/actions/`, especially when input validation, shell safety, or contract compatibility matters. +description: Use when creating or modifying a reusable GitHub composite action under `.github/actions/`, especially when input validation, shell safety, `outputs:` contract compatibility, or backward compatibility matters. Do not use for workflow-level authoring or reuse-pattern selection; route those to internal-github-actions. --- # GitHub Composite Action Skill @@ -12,7 +12,7 @@ for every composite action edit; load only the owner proved by the reusable unit, extracted script depth, or broader workflow question. - `internal-github-actions`: GitHub Actions workflow umbrella and reuse-pattern selection. -- `internal-bash-script`: extracted shell scripts inside composite actions. +- `internal-github`: route back under material routing uncertainty when the owning GitHub lane is unclear or the work is still strategic platform framing. ## When to use diff --git a/.github/skills/internal-github-actions/SKILL.md b/.github/skills/internal-github-actions/SKILL.md index 6588e902..5da21ce4 100644 --- a/.github/skills/internal-github-actions/SKILL.md +++ b/.github/skills/internal-github-actions/SKILL.md @@ -1,6 +1,6 @@ --- name: internal-github-actions -description: Use when authoring or revising GitHub Actions workflows, reusable workflows, or deciding when shared step logic should move into a composite action. +description: Use when authoring or revising GitHub Actions workflows under `.github/workflows/`, creating reusable workflows via `workflow_call`, deciding whether shared step logic should stay inline, move to a reusable workflow, or move to a composite action. Do not use for ruleset, permission, OIDC, or guardrail design; route those to internal-github-governance. --- # GitHub Actions Skill @@ -13,7 +13,7 @@ reuse decision, delivery question, or infrastructure coupling. - `internal-devops-core-principles`: delivery-system strategy, release safety, operational readiness, and incident learning. - `internal-github-action-composite`: composite-action authoring and step-level reuse. -- `internal-terraform`: Terraform resources deployed by CI/CD. +- `internal-github`: route back under material routing uncertainty when the owning GitHub lane is unclear or the work is still strategic platform framing. ## When to use diff --git a/.github/skills/internal-github-governance/SKILL.md b/.github/skills/internal-github-governance/SKILL.md index f91e6fe4..79b057b8 100644 --- a/.github/skills/internal-github-governance/SKILL.md +++ b/.github/skills/internal-github-governance/SKILL.md @@ -1,19 +1,16 @@ --- name: internal-github-governance -description: Use when the user needs GitHub governance guidance for rulesets, branch protection, permissions models, GitHub Apps permissions, Actions permissions, OIDC posture, secret handling, environments, Copilot governance, or other controls that define what actors can do after the GitHub operating model is chosen. +description: Use when the user needs GitHub guardrail or permission guidance — rulesets, branch protection, repository or organization permissions, GitHub Apps permissions, Actions permissions, OIDC posture, secret handling, environments, or Copilot governance — after the broad operating model is known. Do not use for enterprise, org, or repo-model decision framing. --- # Internal GitHub Governance -## Referenced skills - -- `internal-github-strategic`: route back when enterprise, org, repo, or operating-model direction is still unsettled. -- `internal-github-operations`: route runner health, reporting, audit evidence, drift checks, or post-rollout validation. - Use this skill when the next need is to define or review GitHub permissions, guardrails, and policy decisions. This skill owns governance logic after the broad operating model is known. It helps separate enterprise or org guardrails from repo or environment grants and keeps permission decisions auditable. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-github`. + ## When to use - The user needs ruleset, branch protection, or repository permission guidance. @@ -21,12 +18,6 @@ This skill owns governance logic after the broad operating model is known. It he - The user needs secret, environment, or Copilot governance guidance. - The user needs a review of guardrail design, exceptions, or access governance. -## When not to use - -- The main problem is enterprise, org, repo, or mono-repo versus multi-repo decision framing. -- The main problem is runner health, reporting, audit evidence, or post-rollout validation. -- The task is implementation-only. - ## Main domains covered - rulesets and branch protection @@ -72,13 +63,6 @@ For broader asks, return: - exception or blast-radius note - what should be validated before rollout -## Relationship to adjacent skills - -- `internal-github-strategic` - Use first when the user still needs option framing, repo-model tradeoffs, or operating-model decisions. -- `internal-github-operations` - Use when the next need is preflight, audit evidence, drift checks, or validation after the governance design is chosen. - ## Common mistakes | Mistake | Why it matters | Instead | @@ -96,5 +80,5 @@ For broader asks, return: - Confirm the recommended mechanism is clear about whether it prevents, grants, or constrains automation. - Confirm trust boundaries are explicit for Apps, Actions, OIDC, secrets, or Copilot policy choices. - Confirm staged rollout validation is named before high-blast-radius ruleset, permission, or environment changes. -- Confirm the answer says when operational proof should move to `internal-github-operations`. +- Confirm out-of-scope needs, such as operational proof or operating-model framing, are identified as outside this lane instead of being answered here. - Before enabling CODEOWNERS-backed review enforcement in a consumer repository, confirm no template placeholder owner remains. diff --git a/.github/skills/internal-github-operations/SKILL.md b/.github/skills/internal-github-operations/SKILL.md index a84b90b9..c35817e4 100644 --- a/.github/skills/internal-github-operations/SKILL.md +++ b/.github/skills/internal-github-operations/SKILL.md @@ -1,19 +1,16 @@ --- name: internal-github-operations -description: Use when the user needs GitHub operational guidance for Actions health, runner operations, audit logs, reporting and export, drift checks, preflight checks, post-rollout validation, or operational evidence after a governance or operating-model decision has already been made. +description: Use when the user needs GitHub operational guidance — Actions health, runner operations, audit logs, reporting and export, drift checks, preflight checks, post-rollout validation, or operational evidence — after a governance or operating-model decision has already been made. Do not use for direction framing or guardrail design. --- # Internal GitHub Operations -## Referenced skills - -- `internal-github-strategic`: route back when enterprise, org, repo, or operating-model direction is still unsettled. -- `internal-github-governance`: route rulesets, permissions, Actions permissions, OIDC, secrets, environments, or guardrail design. - Use this skill when the next need is to validate, observe, or operationalize a GitHub platform decision. This skill owns the operational side of the platform: workflow health, runner evidence, preflight, and post-rollout verification. It does not replace strategic framing or governance design. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-github`. + ## When to use - The user needs operational readiness guidance after a design choice. @@ -21,12 +18,6 @@ This skill owns the operational side of the platform: workflow health, runner ev - The user needs preflight, post-rollout validation, reporting, or export patterns. - The user needs drift checks or operational proof that a change behaved as expected. -## When not to use - -- The main problem is still choosing the high-level direction. -- The main problem is repo-model, Apps strategy, or governance design. -- The task is a narrow implementation change with no operational design question. - ## Main domains covered - Actions health and workflow evidence @@ -69,13 +60,6 @@ For broader asks, return: - runner, audit, or continuity note when relevant - open operational risks -## Relationship to adjacent skills - -- `internal-github-strategic` - Use first when the core decision is still unsettled. -- `internal-github-governance` - Use when the operations question is actually about rulesets, permissions, OIDC, secret posture, or guardrail design rather than validation. - ## Common mistakes | Mistake | Why it matters | Instead | @@ -83,7 +67,7 @@ For broader asks, return: | Treating workflow success as proof that permissions or guardrails are correct | A single green run can hide excessive privilege or missing failure paths | Check expected permission boundaries, audit trails, and negative cases as separate signals | | Skipping preflight for high-blast-radius rollout | Ruleset, token, runner, or environment regressions surface too late | Define rollout unit, preflight checks, rollback trigger, and owner before rollout | | Reporting only intended policy without operational evidence | Governance looks correct on paper without proof that delivery still works | Record what workflows, runners, and audit surfaces actually showed | -| Mixing validation advice with new governance design instead of keeping the boundary clear | The operations skill stops being a reliable validation owner | Keep new guardrail design in `internal-github-governance` and validate it here | +| Mixing validation advice with new governance design instead of keeping the boundary clear | The operations skill stops being a reliable validation owner | Keep new guardrail design out of the validation answer and validate the chosen design here | | Giving a continuity answer without making the build or release criticality assumption visible | Continuity guidance can be overbuilt or incomplete | State the assumed pipeline or release criticality before recommending the evidence path | | Treating one successful rollout wave as proof for all repositories or environments | Wider repo sets or runner groups can still fail differently | Validate the first safe unit and widen only after recording real evidence | diff --git a/.github/skills/internal-github-pr/SKILL.md b/.github/skills/internal-github-pr/SKILL.md index 9199d52c..a45b0672 100644 --- a/.github/skills/internal-github-pr/SKILL.md +++ b/.github/skills/internal-github-pr/SKILL.md @@ -1,22 +1,13 @@ --- name: internal-github-pr -description: Use when creating, updating, validating, or merging GitHub pull requests in this repository, including PR template bodies, approval checks, merge method, terminal state verification, or PR lifecycle evidence. +description: Use when creating, updating, validating, or merging GitHub pull requests in this repository — PR template bodies, approval and required-review checks, merge method choice, terminal-state verification via `gh pr view --json state,mergedAt`, or PR lifecycle evidence. Do not use for workflow authoring touched by a PR. --- # Internal GitHub PR -## Referenced skills +Owns pull request lifecycle work in this repository: template-compliant bodies, readiness checks, merge method choice, and terminal-state evidence. -This index lists every other skill that this file asks the agent to load, route to, compare against, or delegate to. -Treat the referenced skills below as on-demand supports. Do not preload them -for every PR task; load only the owner proved by the touched surface, review -need, workflow change, or lifecycle claim. - -- `internal-review-code`: defect-first review after PR body or lifecycle work. -- `internal-github-actions`: workflow/action pinning and Actions security rules for PRs that touch workflows. -- `internal-review-high-level`: systems-level impact analysis that feeds PR risk. -- `openai-gh-address-comments`: review-thread remediation after PR body work exists. -- `superpowers-verification-before-completion`: evidence gate before claiming PR readiness, validity, mergeability, or completion. +If the request falls outside this lane, or routing is unclear under material routing uncertainty, route back to `internal-github`. ## When to use @@ -41,9 +32,8 @@ need, workflow change, or lifecycle claim. - Prefer `gh pr merge --squash` over the default merge-commit path unless the repository clearly standardizes on another allowed merge method. - Use `--admin` only when policy explicitly allows a bypass. - Treat organization-wide `gh search prs` results as eventually consistent immediately after merge; confirm terminal state with repository-scoped `gh pr view --json state,mergedAt` before treating a just-merged PR as still open. -- When the PR touches GitHub Actions workflow/action pinning, follow `internal-github-actions` for full-SHA and adjacent release-reference rules. -- Use `superpowers-verification-before-completion` before claiming a PR is ready, - valid, mergeable, merged, or complete. +- When the PR touches GitHub Actions workflow or action pinning, require full-SHA pinning and consistent release references in the PR evidence. +- Verify fresh evidence before claiming a PR is ready, valid, mergeable, merged, or complete. ## Template resolution @@ -96,15 +86,6 @@ If the user provides a specification, issue, or acceptance outline: | Listing every changed file instead of summarizing | Noisy description that obscures intent | Group changes by purpose; detail only non-obvious changes | | Not including validation commands and output | Reviewer has no confidence that code was tested | Always include the exact commands and their results | -## Cross-references - -- **internal-review-high-level**: for systems-level impact analysis that feeds the risk section. -- **internal-review-code**: for the review that follows the PR. -- **internal-github-actions**: for workflow/action pinning and Actions security rules touched by the PR. -- **openai-gh-address-comments**: for addressing review threads and PR comments after the PR body exists; keep review-thread remediation separate from PR lifecycle/body work. -- **superpowers-verification-before-completion**: for evidence before readiness, - mergeability, merge, or completion claims. - ## Validation - Every template-defined section heading is present. @@ -113,5 +94,4 @@ If the user provides a specification, issue, or acceptance outline: - Final PR body is persisted when tooling supports PR updates. - Merge readiness is based on PR-scoped checks and qualifying review evidence. - Recently merged PR state is confirmed with repository-scoped `gh pr view --json state,mergedAt`. -- `superpowers-verification-before-completion` was applied before claiming PR - readiness, validity, mergeability, merge, or completion. +- Fresh evidence was verified before claiming PR readiness, validity, mergeability, merge, or completion. diff --git a/.github/skills/internal-github-strategic/SKILL.md b/.github/skills/internal-github-strategic/SKILL.md deleted file mode 100644 index d820fdc8..00000000 --- a/.github/skills/internal-github-strategic/SKILL.md +++ /dev/null @@ -1,169 +0,0 @@ ---- -name: internal-github-strategic -description: Use when the user needs high-level GitHub platform or operating-model decision support before implementation, and the next step is not yet governance, operations, or delivery work. ---- - -# Internal GitHub Strategic - -## Referenced skills - -- `internal-github-governance`: route when rulesets, permissions, OIDC, secrets, environments, or guardrails become the decision. -- `internal-github-operations`: route when Actions health, runner operations, audit logs, drift, or evidence becomes the decision. -- `internal-github-actions`: route GitHub Actions workflow implementation. -- `internal-python-script`: route implementation work in standalone Python automation. -- `internal-bash-script`: route implementation work in standalone Bash automation. - -Use this skill when the main need is to reason about a GitHub platform decision before implementation. - -This is a strategic support skill. It helps frame the decision, compare realistic options, expose tradeoffs, and recommend a direction. It does not implement the change and it does not choose Terraform, Python, or Bash on behalf of the user. - -This skill absorbs light structure concerns such as enterprise, organization, and repository model choices. Do not force a separate structure lane unless a future boundary clearly emerges. - -## When to use - -- The user needs GitHub decision support before execution. -- Multiple GitHub approaches are credible and tradeoffs matter. -- The user wants a recommendation grounded in current GitHub guidance. -- The user wants high-level support for enterprise, org, repo, Apps, Actions, runner, Copilot, or spend decisions. - -## When not to use - -- The task is already a clear implementation change. -- The user only needs detailed ruleset, permission, OIDC, secret, or runner-operation implementation. -- The task is purely post-rollout validation or evidence gathering. -- The request is narrow and operational with no real decision to frame. - -## Main domains covered - -- enterprise, organization, and repository model decision framing -- mono-repo versus multi-repo tradeoffs when relevant -- GitHub Apps strategy -- GitHub Actions strategy -- runner strategy at decision level -- Copilot direction and governance implications at decision level -- licensing, spend, and cost-value implications at decision level -- operational and governance implications at decision level - -## Optional lens activation - -Do not load every lens by default. - -Use only the minimum set of lenses needed for the request. If the user explicitly names one or more lenses, prioritize only those. If the user does not name lenses, infer the smallest useful set. - -Available lenses include: - -- security -- identity and access -- governance -- operations -- runner model -- Copilot -- BC/DR -- FinOps -- compliance -- rollout and rollback -- blast radius -- maintainability - -Rules: - -- Start narrow. -- Expand only when the request is broad, risky, or ambiguous. -- If another lens would materially improve the recommendation, suggest it briefly instead of forcing it. -- Keep the active lenses explicit when more than one is in play. - -Load `references/lens-playbook.md` when the user wants a deeper framing aid or when the choice of lenses is not obvious. - -## Optional BC/DR lens - -BC/DR is optional. - -Activate it only when: - -- the user asks about delivery continuity, runner resilience, backup, recovery, or failover expectations -- the decision has clear continuity implications for build, release, or repository operations -- the recommendation would be materially incomplete without it - -If BC/DR seems relevant but is not requested, suggest it as an optional lens instead of forcing it. - -## Use of current documentation - -Use current GitHub documentation only when freshness materially affects the answer, especially for GitHub Apps permissions, Actions behavior, runner support, OIDC guidance, rulesets, or Copilot product boundaries. - -Do not invoke current-doc research by default for stable, generic reasoning. - -## Mandatory behavior - -- Identify the decision first, not the implementation tool. -- Make assumptions explicit. -- Compare realistic options, not strawmen. -- Keep tradeoffs concrete. -- Surface material risk, blast radius, and reversibility when relevant. -- Include cost-value considerations when they matter to the decision. -- Stay proportional to the size of the question. - -## Adaptive output modes - -Choose the lightest output that fits the request. - -### Quick answer - -Use for narrow asks. - -Include: - -- direct recommendation -- short rationale -- optional risk or follow-up note - -### Decision note - -Use for normal strategic support. - -Include: - -- decision statement -- key options or tradeoff -- recommended direction -- main risk or validation note - -### Deep analysis - -Use only for broad, ambiguous, high-risk, or explicitly detailed requests. - -Include: - -- context and assumptions -- options considered -- active lenses used -- recommendation and why it wins -- main risks and blast radius -- validation or follow-up path - -## Relationship to adjacent skills - -- `internal-github-governance` - Use when the next need is rulesets, permissions, OIDC, secret posture, environments, or Copilot guardrail definition. -- `internal-github-operations` - Use when the next need is Actions health, runner validation, audit evidence, reporting, drift checks, or post-rollout checks. -- `internal-github-actions`, `internal-python-script`, `internal-bash-script` - Use when the decision is settled and implementation begins. - -## Common mistakes - -| Mistake | Why it matters | Instead | -| --- | --- | --- | -| Forcing a full multi-lens analysis for a small question | The answer gets heavy without improving the decision | Start with the smallest useful lens set and widen only if risk or ambiguity justifies it | -| Treating BC/DR as mandatory for every answer | Continuity concerns can crowd out the actual GitHub operating-model choice | Activate BC/DR only when build, release, or repository continuity materially changes the recommendation | -| Recommending a direction without current-source verification when freshness matters | Product boundaries, permission models, or licensing limits may have changed | Call out the freshness dependency and say which GitHub fact still needs current verification | -| Confusing decision support with implementation guidance | The user loses the strategic framing they asked for | Keep the answer at decision level and hand off only after the direction is chosen | -| Expanding into tool selection when the user did not ask for it | The response drifts from GitHub tradeoffs into execution detail | Keep the recommendation centered on the platform choice, not the delivery tooling | -| Forcing GitHub into a cloud-provider structure pattern when the boundary is weaker | The strategic lane gets distorted and a fake structure owner pressure appears | Keep light enterprise, org, and repo-shape decisions inside strategic unless a real boundary emerges | - -## Validation - -- Confirm the decision statement is explicit and narrow enough that the next owner is obvious. -- Confirm assumptions, active lenses, and the main tradeoff are named instead of implied. -- Confirm the recommendation includes reversibility or blast-radius guidance when the choice is hard to unwind. -- Confirm viable options are compared in GitHub-local terms such as repo model, Apps trust, runners, or Copilot posture. -- Confirm the answer states when freshness matters and which current GitHub fact still needs validation. diff --git a/.github/skills/internal-github-strategic/agents/openai.yaml b/.github/skills/internal-github-strategic/agents/openai.yaml deleted file mode 100644 index 88e0068a..00000000 --- a/.github/skills/internal-github-strategic/agents/openai.yaml +++ /dev/null @@ -1,4 +0,0 @@ -interface: - display_name: "Internal GitHub Strategic" - short_description: "GitHub decision framing and tradeoff support" - default_prompt: "Use $internal-github-strategic to frame this GitHub platform decision, keep the answer proportional, and activate only the lenses that matter." diff --git a/.github/skills/internal-github/SKILL.md b/.github/skills/internal-github/SKILL.md new file mode 100644 index 00000000..d755930b --- /dev/null +++ b/.github/skills/internal-github/SKILL.md @@ -0,0 +1,182 @@ +--- +name: internal-github +description: Use when a GitHub task cannot be routed confidently to a specific GitHub skill because the request is materially ambiguous, has multiple GitHub domains with no clear primary owner, or requires clarification before selecting the correct specialist, or when the user needs high-level GitHub platform or operating-model decision support or tradeoff framing before implementation. Do not use for clearly scoped governance, operations, PR lifecycle, Actions workflow authoring, composite-action authoring, or current Copilot platform behavior research. +--- + +# Internal GitHub + +Fallback router for GitHub tasks that cannot be assigned confidently to one specialist, and strategic support skill for high-level GitHub platform and operating-model decision framing. Do not activate only because the task concerns GitHub; activate only when material routing uncertainty blocks owner selection or when the user needs decision support before the next step is governance, operations, or delivery. + +## Referenced skills + +- `internal-github-governance`: rulesets, branch protection, repository and organization permissions, GitHub Apps permissions, Actions permissions, OIDC posture, secrets, environments, Copilot governance. +- `internal-github-operations`: Actions health, runner operations, audit logs, reporting, drift checks, preflight checks, post-rollout validation, operational evidence. +- `internal-github-actions`: GitHub Actions workflow authoring under `.github/workflows/`, reusable workflows, reuse-pattern selection. +- `internal-github-action-composite`: composite-action authoring under `.github/actions/`, input validation, shell safety, contract compatibility. +- `internal-github-pr`: PR creation, body, merge readiness, merge method, terminal-state verification, PR lifecycle evidence. +- `internal-copilot-docs-research`: current GitHub Copilot or MCP platform behavior research when freshness materially affects the answer. + +## When to use + +- Material ambiguity prevents selecting one primary GitHub specialist. +- Multiple GitHub domains are material and no primary owner can be identified safely. +- The user explicitly invokes `$internal-github`. +- The task asks which GitHub lane should own the work before requesting a domain solution. +- The user needs high-level GitHub platform or operating-model decision support or tradeoff framing before implementation. + +## Routing threshold + +Activate only when at least one holds: + +- the request is materially ambiguous and clarification is required before a GitHub owner can be selected; +- multiple GitHub domains are material and no primary owner can be identified safely; +- the task asks which GitHub problem-solving lane should own the work; +- the user needs strategic decision framing and the next step is not yet governance, operations, or delivery. + +Do not activate when one specialist clearly owns the next step; route directly to that specialist instead. Explicit `$internal-github` invocation remains valid. + +## Handoffs + +| To | Owns | +|---|---| +| `internal-github-governance` | rulesets, branch protection, repo and org permissions, GitHub Apps permissions, Actions permissions, OIDC, secrets, environments, Copilot governance | +| `internal-github-operations` | Actions health, runner operations, audit logs, reporting, drift, preflight, post-rollout validation, evidence | +| `internal-github-actions` | workflow authoring under `.github/workflows/`, reusable workflows, reuse-pattern selection | +| `internal-github-action-composite` | composite-action authoring under `.github/actions/`, input validation, shell safety, contract compatibility | +| `internal-github-pr` | PR creation, body, merge readiness, merge method, terminal-state verification, PR lifecycle evidence | +| `internal-copilot-docs-research` | current GitHub Copilot or MCP platform behavior when freshness materially affects the answer | + +## Dispatch contract + +1. State the routing uncertainty. +2. Identify the candidate GitHub owners. +3. Select the minimum specialist set. +4. Keep strategic comparison here only while it is needed to choose the owner. +5. Hand the resolved task to the primary specialist instead of retaining ownership. + +## On-demand references + +Load `references/routing-matrix.md` for the routing decision tree. Load `references/strategic-framing.md` when the choice of GitHub owner or lens needs worked lens combinations, decision-note depth, or worked-shape comparison before handoff. Do not load either by default for a clearly scoped single-owner request. + +## Optional lens activation + +Do not load every lens by default. + +Use only the minimum set of lenses needed for the request. If the user explicitly names one or more lenses, prioritize only those. If the user does not name lenses, infer the smallest useful set. + +Available lenses include: + +- security +- identity and access +- organization and repo model +- governance +- operations +- runner model +- Copilot +- BC/DR +- FinOps +- compliance +- rollout and rollback +- blast radius +- maintainability + +Rules: + +- Start narrow. +- Expand only when the request is broad, risky, or ambiguous. +- If another lens would materially improve the recommendation, suggest it briefly instead of forcing it. +- Keep the active lenses explicit when more than one is in play. + +## Optional BC/DR lens + +BC/DR is optional. + +Activate it only when: + +- the user asks about delivery continuity, runner resilience, backup, recovery, or failover expectations +- the decision has clear continuity implications for build, release, or repository operations +- the recommendation would be materially incomplete without it + +If BC/DR seems relevant but is not requested, suggest it as an optional lens instead of forcing it. + +## Use of current documentation + +Use current GitHub documentation only when freshness materially affects the answer. When the question is about Copilot or MCP behavior specifically, route to `internal-copilot-docs-research` instead of answering from memory. + +## Mandatory behavior + +- Identify the decision first, not the implementation tool. +- Make assumptions explicit. +- Compare realistic options, not strawmen. +- Keep tradeoffs concrete. +- Surface material risk, blast radius, and reversibility when relevant. +- Include cost-value considerations when they matter to the decision. +- Stay proportional to the size of the question. + +## Adaptive output modes + +Choose the lightest output that fits the request. + +### Quick answer + +Use for narrow asks. + +Include: + +- direct recommendation +- short rationale +- optional risk or follow-up note + +### Decision note + +Use for normal strategic support. + +Include: + +- decision statement +- key options or tradeoff +- recommended direction +- main risk or validation note + +### Deep analysis + +Use only for broad, ambiguous, high-risk, or explicitly detailed requests. + +Include: + +- context and assumptions +- options considered +- active lenses used +- recommendation and why it wins +- main risks and blast radius +- validation or follow-up path + +## Anti-scope + +- Do not use this fallback for a clearly scoped ruleset, permission, OIDC, secret, environment, or Copilot governance request. Route directly to `internal-github-governance`. +- Do not use this fallback for Actions health, runner operations, audit log, drift, preflight, or post-rollout validation. Route directly to `internal-github-operations`. +- Do not use this fallback for workflow authoring, reusable workflow design, or reuse-pattern selection. Route directly to `internal-github-actions`. +- Do not use this fallback for composite-action authoring or contract changes under `.github/actions/`. Route directly to `internal-github-action-composite`. +- Do not use this fallback for PR creation, body, merge readiness, or lifecycle evidence. Route directly to `internal-github-pr`. +- Do not use this fallback when the only unresolved question is current Copilot or MCP platform behavior. Route directly to `internal-copilot-docs-research`. + +## Anti-patterns + +- Activating this fallback when one specialist clearly owns the next step. +- Forcing a full multi-lens analysis for a small question. +- Answering a governance, operations, or workflow question here instead of handing it to the specialist. +- Keeping ownership after the lane is resolved instead of handing off. +- Recommending implementation tooling when the user only asked for routing. +- Invoking current-doc research by default for stable, generic reasoning. + +## Validation + +- State why the request could not be assigned to one primary GitHub specialist. +- Confirm the selected specialist set is the minimum needed to resolve the uncertainty. +- Confirm the resolved task is handed to a primary specialist and not retained by this fallback. +- Confirm assumptions, tradeoffs, and the next owner are explicit. +- Confirm lenses, when used, are the minimum set and named explicitly when more than one is active. +- Confirm the decision statement is explicit and narrow enough that the next owner is obvious. +- Confirm the recommendation includes reversibility or blast-radius guidance when the choice is hard to unwind. +- Confirm cost-value or operational impact, including licensing or runner cost, is called out when it materially changes the recommendation. +- Confirm the answer states when freshness matters and whether current GitHub or Copilot behavior still needs verification. diff --git a/.github/skills/internal-github/agents/openai.yaml b/.github/skills/internal-github/agents/openai.yaml new file mode 100644 index 00000000..3398587b --- /dev/null +++ b/.github/skills/internal-github/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "Internal GitHub" + short_description: "GitHub routing and strategic decision support" + default_prompt: "Use $internal-github to route an unclear GitHub task to the minimum specialist set, or to frame a GitHub platform or operating-model decision when the next step is not yet governance, operations, or delivery." diff --git a/.github/skills/internal-github/references/routing-matrix.md b/.github/skills/internal-github/references/routing-matrix.md new file mode 100644 index 00000000..fc13f56e --- /dev/null +++ b/.github/skills/internal-github/references/routing-matrix.md @@ -0,0 +1,32 @@ +# GitHub Routing Scenario Matrix + +## Fallback-positive cases + +| Scenario | Why no primary owner | +|---|---| +| Underspecified cross-domain GitHub problem mixing governance, operations, and workflow authoring with no clear primary deliverable | The request names multiple GitHub domains but does not identify which deliverable takes priority. | +| GitHub platform question asking which lane should own the work without naming governance, operations, actions, composite, PR, or Copilot research | The user has not selected a domain; the fallback must clarify the lane before any specialist can engage. | +| Broad GitHub adoption review where the user wants a general health assessment across all domains | No single specialist owns a cross-domain health review; the fallback selects the minimum set. | + +## Direct-specialist negative cases + +| Scenario | Direct owner | Reason | +|---|---|---| +| Ruleset, branch protection, repository or organization permissions, GitHub Apps permissions, OIDC, secrets, environments, or Copilot governance | `internal-github-governance` | The deliverable is a guardrail or permission boundary. | +| Actions health, runner operations, audit-log review, reporting, drift, preflight, or post-rollout validation | `internal-github-operations` | The deliverable is operational evidence or continuity proof. | +| Workflow authoring under `.github/workflows/` or reusable workflow design | `internal-github-actions` | The deliverable is workflow behavior. | +| Composite action authoring under `.github/actions/` or contract compatibility | `internal-github-action-composite` | The deliverable is the reusable step unit. | +| PR creation, body, merge readiness, merge method, or terminal-state verification | `internal-github-pr` | The deliverable is PR lifecycle evidence. | +| Current GitHub Copilot or MCP platform behavior verification | `internal-copilot-docs-research` | The deliverable is current-source research. | + +## Multi-domain primary-owner cases + +| Scenario | Primary owner | Secondary | Reason | +|---|---|---|---| +| Org or repo-model decision with later ruleset work | `internal-github-governance` | `internal-github-operations` | The first deliverable is placement; ruleset validation follows once the model is settled. | +| Ruleset rollout evidence | `internal-github-governance` | `internal-github-operations` | The first deliverable is governance design; operations validates the rollout. | +| Workflow permission detail | `internal-github-actions` or `internal-github-governance` | depends on deliverable | Choose `internal-github-actions` when the deliverable is workflow behavior; choose `internal-github-governance` when the deliverable is the permission boundary. | + +## Review rule + +Prefer a direct specialist whenever a reasonable reviewer can name one primary owner from the request itself. Activate the fallback only when the request does not identify a primary owner and clarification is required before a specialist can engage. diff --git a/.github/skills/internal-github-strategic/references/lens-playbook.md b/.github/skills/internal-github/references/strategic-framing.md similarity index 94% rename from .github/skills/internal-github-strategic/references/lens-playbook.md rename to .github/skills/internal-github/references/strategic-framing.md index d59951a2..4d5d0f52 100644 --- a/.github/skills/internal-github-strategic/references/lens-playbook.md +++ b/.github/skills/internal-github/references/strategic-framing.md @@ -1,6 +1,6 @@ -# GitHub Strategic Lens Playbook +# GitHub Strategic Framing Reference -Use this reference when the user wants more depth than the base skill should load by default. +Use this reference when the fallback trigger fires and the choice of GitHub owner or lens needs worked combinations, decision-note depth, or worked-shape comparison before handoff. Do not load it for single-owner requests or routine routing. ## Common lens combinations @@ -20,12 +20,6 @@ Use this reference when the user wants more depth than the base skill should loa - The decision changes delivery continuity or recovery posture: suggest `BC/DR` - The choice materially affects developer workflow or repo shape: suggest `maintainability` -## Depth control - -- Stay in `Quick answer` mode when one option is clearly better and the user asked a narrow question. -- Upgrade to `Decision note` when at least two viable options exist. -- Upgrade to `Deep analysis` only when the user asks for it or the risk profile justifies it. - ## Worked GitHub decision shapes ### Enterprise or repo-model choice @@ -71,3 +65,9 @@ Use this when the question is too consequential for a quick answer but does not | The choice does not alter trust, continuity, or repo ownership posture | The choice changes repo model, Apps trust, runner model, or Copilot posture | | The answer can stay within one lens without hiding material risk | A second lens changes the recommendation or the risk statement | | Freshness is not the deciding factor | Current GitHub behavior, permissions, or product limits could change the outcome | + +## Depth control + +- Stay in `Quick answer` mode when one option is clearly better and the user asked a narrow question. +- Upgrade to `Decision note` when at least two viable options exist. +- Upgrade to `Deep analysis` only when the user asks for it or the risk profile justifies it. diff --git a/.github/skills/internal-review-high-level/references/plan-completion-audit.md b/.github/skills/internal-review-high-level/references/plan-completion-audit.md index f4afbad5..489d3221 100644 --- a/.github/skills/internal-review-high-level/references/plan-completion-audit.md +++ b/.github/skills/internal-review-high-level/references/plan-completion-audit.md @@ -2,7 +2,7 @@ Use this reference when review needs to map a retained plan or promised scope to what actually changed. Keep the audit evidence-first. Do not accept a completion -claim from chat memory, a `done-*` marker, or intent alone. +claim from chat memory or intent alone. ## Source Pattern @@ -15,35 +15,31 @@ claim from chat memory, a `done-*` marker, or intent alone. Use the inline completion evidence from the active gateway owner for small, direct tasks. Run this full audit when any condition is true: -- The retained plan has more than 6 executable items or numbered plan files. +- The retained plan has more than 6 executable items. - The diff crosses multiple repository-owned asset families. - The change modifies always-on guidance, wrapper agents, validators, or tests. - A completion claim depends on manual evidence, external state, or a missing validator. -- Numbered plan files were correctly removed by the `done-*` loop and completion - now depends on reconstructed evidence. - A reviewer asks whether the delivered diff really matches the approved plan. -## Evidence Envelope Inputs +## Inputs -When numbered plan files were removed by a correct `done-*` loop, use the -evidence envelope as the plan-to-delivery source. The envelope should reconstruct -each original or completed item and cite the file, diff, artifact, or validator -evidence that supports its status. +The audit requires: -If no envelope exists, reconstruct from `done-*` files and reachable artifacts. -If reconstruction cannot produce item-level evidence, mark the affected item -`UNVERIFIABLE` and downgrade the completion state or route the gap. +- **Exact plan file** — the approved retained plan under `tmp/superpowers/plans/`. +- **Exact status sibling** — the `..md` file produced by the executor. +- **Matching fingerprint** — the SHA-256 hash in the status file must match the plan file. + +Use the status sibling's `## Completed`, `## Remaining`, and `## Validation` sections as the item-level evidence source. ## Status Vocabulary | Status | Criteria | Route | | --- | --- | --- | -| `DONE` | The item has direct evidence in the diff, target file, or validator output. | Keep as completed. | +| `DONE` | Every required item has direct evidence in the diff, target file, or validator output. | Keep as completed. | | `PARTIAL` | Some required behavior shipped, but a named subpart remains absent or unverified. | Route the missing part to delivery or defer with risk. | -| `NOT_DONE` | No credible evidence shows the item was implemented. | Route to delivery or mark as an explicit non-action. | -| `CHANGED` | The item was intentionally satisfied by a different approach that still meets the approved target. | Cite the replacement and validate scope fit. | -| `UNVERIFIABLE` | The item cannot be checked from available files, diff, reachable paths, or validator output. | State the evidence gap and request proof or downgrade completion state. | +| `NEEDS_REVIEW` | Execution is complete but a human or external verification remains. | State the evidence gap and request proof. | +| `BLOCKED` | A named blocker prevents further execution. | Route to delivery or mark as an explicit non-action. | ## Verification Classes @@ -56,23 +52,20 @@ Classify each plan item before judging status: - `EXTERNAL_VERIFIABLE`: another repository or service must be inspected. If an item is `MANUAL_VERIFIABLE` or `EXTERNAL_VERIFIABLE` and no evidence is -available, mark it `UNVERIFIABLE` instead of guessing. +available, mark it `NEEDS_REVIEW` instead of guessing. ## Procedure -1. Extract every executable item from the plan. Use `02-control.md` - (`extended`) or `02-execution.md` (`compact`) when present, and ignore - `questions.md`. -2. If numbered plan files are absent because the `done-*` loop removed them, - use the evidence envelope or reconstruct items from `done-*` files and the - source-item ledger. -3. Record the declared target state, anti-scope, owner, and validation path. -4. Collect observed evidence from changed files, `git diff`, validators, and +1. Verify the status sibling fingerprint matches the plan file. +2. Extract every executable item from the plan. +3. Read the `## Completed`, `## Remaining`, and `## Validation` sections from the status sibling. +4. Record the declared target state, anti-scope, owner, and validation path. +5. Collect observed evidence from changed files, `git diff`, validators, and reachable target paths. -5. Assign a verification class and status to each item. -6. Route every `PARTIAL`, `NOT_DONE`, and `UNVERIFIABLE` item to delivery, +6. Assign a verification class and status to each item. +7. Route every `PARTIAL` and `NEEDS_REVIEW` item to delivery, planning, critical challenge, or defer. -7. Re-check whether the observed delivery introduced scope drift with +8. Re-check whether the observed delivery introduced scope drift with `scope-drift.md`. ## Output Table Template @@ -81,13 +74,10 @@ available, mark it `UNVERIFIABLE` instead of guessing. | --- | --- | --- | --- | --- | | `` | `DIFF_VERIFIABLE` | `DONE` | `` | `` | -## `UNVERIFIABLE` Rules +## `NEEDS_REVIEW` Rules -- Missing plan files, unreadable paths, missing validators, and external-only - claims are `UNVERIFIABLE` unless independent evidence exists. -- A `done-*` file is a progress marker, not proof. Re-open the item when the - target file or validator evidence is absent. -- An evidence envelope may replace removed numbered plan files only when it - preserves item, status, evidence, and route. -- Do not mark `SHIPPED` in a completion report while any required item remains - `UNVERIFIABLE` without an explicit accepted risk. +- Missing validators, unreadable paths, and external-only + claims are `NEEDS_REVIEW` unless independent evidence exists. +- Do not mark `DONE` while any required item remains + `NEEDS_REVIEW` without an explicit accepted risk. +- `DONE` only when every required item is evidence-backed. diff --git a/.github/skills/internal-skill-creator/SKILL.md b/.github/skills/internal-skill-creator/SKILL.md index 8d046e8f..a5ba0a00 100644 --- a/.github/skills/internal-skill-creator/SKILL.md +++ b/.github/skills/internal-skill-creator/SKILL.md @@ -1,206 +1,72 @@ --- name: internal-skill-creator -description: Use first when creating, splitting, replacing, or materially revising a repository-owned skill under `.github/skills/`, especially when trigger, boundary, or validation decisions must be made locally before `openai-skill-creator`. +description: Use when creating or materially revising repository-owned skills under `.github/skills/`, including splits, replacements, or changes to scope, triggers, structure, or validation. --- # Internal Skill Creator -## Referenced skills +## Core method -This index lists every other skill that this file asks the agent to load, route to, compare against, or delegate to. Keep it current before changing downstream wording. Self-references to `internal-skill-creator` identify this current skill and are not a dependency. +`/mattpocock-writing-great-skills` is the core method for skill authoring and +revision. Load it before drafting. Apply its relevant rules throughout the +change instead of repeating them here. -- `openai-skill-creator`: generic bundle anatomy, reusable-resource layout, `agents/openai.yaml`, initialization workflow, and structural validation after the local boundary is clear. -- `local-agent-sync-external-resources`: catalog-governance operating engine behind the local sync command center when work becomes sync-managed catalog maintenance, external refresh, or inventory-wide keep/update/extract/retire decisions. -- `internal-agent-creator`: repository-owned agent authoring and agent/skill boundary rewrites. +## Cross-skill notation -Use this skill as the canonical repository-owned first entrypoint for skill authoring in this repository. +Prefix every cross-skill invocation with `/`. -Keep the ownership model explicit: - -- `internal-skill-creator` is the canonical local owner for repository-owned `.github/skills/` work. -- `openai-skill-creator` is the core operating engine inside that wrapper for bundle anatomy, reusable resources, `agents/openai.yaml`, initialization workflow, and structural validation. -- This skill adds the repository-specific gate: prove the need, choose reuse versus creation, keep triggers retrieval-safe, harden the result against rationalization and boundary drift, and delegate only the remaining bundle work to OpenAI. - -This means `internal-skill-creator` should trigger first for repository-owned skill work, establish the local boundary, and then deliberately hand only the remaining bundle mechanics to `openai-skill-creator` instead of competing with it or duplicating it. - -Treat skill self-containment as a default local quality gate: when creating or -materially revising a skill, prefer a bundle that can be understood, copied, -and validated from its own directory without depending on instructions, -examples, or the only runnable automation living elsewhere in the repository. - -Use `local-agent-sync-external-resources` through `local-sync-external-resources` when the task is broader catalog governance, sync-managed external assets, or inventory-wide retirement and refresh work. - -Use `internal-agent-creator` when the primary output is an agent change or an agent/skill boundary rewrite. - -## Read first - -- Read the target `SKILL.md` plus the nearest competing skills that could already own the request. -- Read root `AGENTS.md` and `.github/copilot-instructions.md` before changing repository-owned scope or policy language. -- Read `.github/INVENTORY.md` when a skill may be added, retired, renamed, or replaced. -- Load `references/writing-skills-checklist.md` when creating a new skill or materially revising an existing one. -- Read `openai-skill-creator` only after the local boundary is clear and only for the remaining bundle work that this skill is not meant to repeat. -- Confirm the target `SKILL.md` has an up-to-date `## Referenced skills` section immediately after the H1 before finalizing a new or materially revised skill. -- Inventory the touched bundle itself before widening scope: `SKILL.md`, - `agents/openai.yaml`, existing `references/`, `scripts/`, `assets/`, and - `fixtures/` if present. +Use the bare `skill-name` when a skill is only named or referenced, including +reference lists, identifiers, state labels, fixtures, scripts, and catalog +entries. Use `/skill-name` whenever an operational verb asks the agent to +load, run, use, invoke, delegate to, or route work to that skill. Apply this +distinction consistently to the six internal gateway skills covered by this +convention. Keep the target skill model-invocable; a called skill must not set +`disable-model-invocation: true`. ## When to use -- Creating a new repository-owned skill under `.github/skills/`. -- Replacing or splitting an existing repository-owned skill whose current boundary is wrong. -- Materially revising a repository-owned skill's scope, trigger, structure, bundled resources, or validation. -- Tightening a skill whose description is too broad, too procedural, or too weak to retrieve reliably. - -## When not to use - -- The task is catalog governance, inventory maintenance, or sync routing. Use `local-agent-sync-external-resources` through `local-sync-external-resources` instead. -- The task is primarily agent authoring or agent/skill architecture. Use `internal-agent-creator` instead. -- The task is outside `.github/skills/` or does not change a repository-owned skill. -- The existing skill already covers the need and the change is a pure copyedit that does not affect retrieval, boundary, validation, or bundle structure. +- The requested skill change affects repository-owned behavior or structure. -## Division of labor +## Local reference -Treat `internal-skill-creator` and `openai-skill-creator` as complementary, not symmetric. +Read `references/authoring-and-evaluation.md` when creating a skill, changing +its boundary or trigger, or selecting an evaluation branch. -This skill should do: - -- decide whether the task really belongs to a repository-owned skill in this repository -- choose no-op, reuse, revise in place, split, replace, or retire -- define the local ownership boundary and trigger wording -- enforce bundle self-containment for created or materially revised skills -- enforce the failing-baseline rule, token discipline, and skill-type testing expectations -- re-check routing fallout in nearby repository-owned assets - -`openai-skill-creator` should do: - -- scaffold or normalize the bundle shape -- handle reusable-resource anatomy for `references/`, `scripts/`, `assets/`, and `agents/openai.yaml` -- provide initializer, metadata-generation, and structural-validation workflow - -Do not restate the full OpenAI creation workflow here. Use this skill to decide and constrain the work, then hand off only the remainder that OpenAI already handles better. - -## Trigger precedence - -- For any repository-owned skill work under `.github/skills/`, start with `internal-skill-creator`, not `openai-skill-creator`. -- Use `openai-skill-creator` only after this wrapper has established that the task really belongs to a repository-owned skill in this repository and there is remaining bundle work that should not be duplicated locally. -- If both skills appear relevant, prefer this skill first because its description is the repo-local route and the OpenAI skill is the embedded engine. -- If `openai-skill-creator` is already in play for generic bundle mechanics, hand repository-specific routing, ownership, and retrieval decisions back to this skill instead of letting OpenAI guess the local policy. +## Workflow -## Decision gate +### 1. Repository preflight -| Situation | Best answer | -| --- | --- | -| Wording cleanup with no change to retrieval, owner, or validation | Update in place or do nothing | -| Same owner, but weak trigger/body/validation is causing misses | Revise the existing skill | -| One skill is handling two intents or colliding with another local owner | Split, replace, or retire the weaker skill | -| The change affects multiple skills, inventory meaning, or sync-managed assets | Use `local-agent-sync-external-resources` through `local-sync-external-resources` | -| The local decision is made and the remaining work is bundle anatomy, reusable resources, `agents/openai.yaml`, or validator usage | Delegate that remainder to `openai-skill-creator` | +Read the target `SKILL.md`, the nearest competing skills, and the applicable +`AGENTS.md`. Inventory the touched bundle. Read `.github/INVENTORY.md` only +when adding, retiring, renaming, or replacing a skill. -## Core rules +Completion criterion: the intended boundary, anti-scope, touched files, and +repository validation path are explicit. -- Start by checking whether an existing repository-owned skill can be reused, narrowed, or updated in place. -- Do not create a new skill until you can state the concrete failure, ambiguity, or repeated authoring miss it must prevent. -- Require a baseline failure before a new or materially revised skill is accepted. If the undesired behavior has not been observed, the case is not ready. -- Check frontmatter integrity before debating trigger wording. Broken frontmatter is a structural failure, not a content-polish issue. -- Review the nearest competing skills before editing. Retrieval quality is judged against neighboring owners, not in isolation. -- Prefer the smallest change that fixes the local problem. -- Make descriptions searchable with concrete terms people would actually type: skill, trigger, `.github/skills/`, `SKILL.md`, create, replace, revise, update, reuse, validation. -- Use active, searchable naming when creating a new skill. Prefer direct verbs or action-shaped names over abstract labels when that improves retrieval. -- A good outcome may be reuse, narrowing, deletion, or replacement. Do not let the workflow bias toward creating another skill. -- Every repository-owned `SKILL.md` this skill creates or materially revises must keep `## Referenced skills` immediately after the H1 as an audit index, not a preload list. -- In `## Referenced skills`, list every other skill the file asks the agent to load, route to, compare against, or delegate to; use `- None.` only when no other skill is referenced. -- Update `## Referenced skills` whenever adding, removing, renaming, or repairing a skill reference. Treat stale or missing skill names as validation failures. -- When a referenced-skill rule changes, verify the touched skill against nearby skills that cite it or that it cites, so the index remains a lazy routing contract instead of drifting into preload wording. -- In `SKILL.md`, reference another skill by name and behavior only. Do not cite file paths inside another skill bundle; those files are private to the owning skill and may change. -- In source-side skill Markdown, cite only paths that exist on disk in the source repository. When sync materializes a target-only file, prefer the source template path or descriptive prose over the consumer-only materialized path. -- When creating or materially revising a skill that introduces scripts, CLIs, or deterministic automation, require an explicit output-contract decision: operator default output, model-facing bounded or compact output, and when full JSON is required for audits. +### 2. Core authoring and revision -For trigger-wording, token-discipline, loophole, and skill-type test detail, enforce `references/writing-skills-checklist.md`. +Load `/mattpocock-writing-great-skills` as the core method. Draft or revise the +smallest coherent bundle. Check invocation, description, information hierarchy, +retrieval quality, and predictability. Remove duplication, sediment, and no-ops; +revise the draft in place instead of only reporting findings. -## OpenAI handoff points +Completion criterion: every applicable core rule is reflected in the draft, +and each retained local instruction has a repository-specific reason to exist. -After the local decision gate is complete, hand off to `openai-skill-creator` only when you need one or more of these: +### 3. Proportional evaluation -- new bundle scaffolding -- regeneration or repair of `agents/openai.yaml` -- bundle-shape guidance for `references/`, `scripts/`, or `assets/` -- structural validation via the OpenAI validator +Read `references/authoring-and-evaluation.md`. Select the applicable evaluation +branches. Record skipped branches and reasons. -## Baseline evidence +Completion criterion: applicable branches have evidence; evidence, blockers, +and completion status are explicit. -- Iron law: no new skill and no material skill edit without a failing baseline first. -- Accept concrete local evidence such as a failed retrieval, repeated review feedback, trigger overlap, weak discovery wording, stale validation expectations, or a documented miss in `tmp/superpowers/`. -- Reject vague justification such as "this feels reusable", "the repo might need it later", or "the text looks light". +### 4. Repository closure -## Workflow +1. Update `agents/openai.yaml` to match the revised skill purpose. +2. Run `python3 .github/scripts/validate_internal_skills.py --skill --strict`. +3. Check routing fallout in nearby skills and agents. +4. Record before/after line and word counts for the touched bundle. -1. Prove the need first. - Record the baseline failure, ambiguity, or repeated authoring miss the skill must prevent. -2. Reject the weakest answer. - Prefer reuse, tightening an existing trigger, or doing nothing when the evidence does not justify a new repository-owned owner. -3. Set the boundary before writing. - Decide what this skill owns locally and which adjacent owner should win when the task is really sync governance, agent authoring, or another domain. -4. Isolate the remainder. - Identify which parts of the job are still local policy work and which parts are now generic OpenAI bundle work. -5. Enforce bundle locality before handoff. - Repair repo-rooted self-references, missing bundle-local examples, and any - external-only operating engine that would break a copied-out skill. -6. Hand off only the remainder that OpenAI already handles well. - Load `openai-skill-creator` for scaffolding, resource anatomy, `agents/openai.yaml`, or structural validation, but do not replay its full workflow in this skill. -7. Resume local control for wrapper checks. - Use `references/writing-skills-checklist.md` to tighten trigger wording, token discipline, loophole closure, and test design. -8. Validate the right thing. - Ensure OpenAI-side structural checks ran if bundle mechanics changed, then check retrieval quality plus skill-type behavior before treating the skill as done. -9. Re-check routing fallout. - Update nearby references or paired agent text only when the visible local entrypoint or ownership meaning actually changed. - -Use `references/writing-skills-checklist.md` for the anti-rationalization rules, token-discipline reminders, and skill-type testing expectations that this wrapper should enforce. - -## Validation - -- Automated: `python3 ./.github/scripts/validate_internal_skills.py --skill ` (covers name match, description trigger-first, openai.yaml shape, local-reference existence, cross-skill-file-reference absence, body line count, inline-template density). Remaining checks below are manual. - -Then confirm: - -- `name:` matches the folder name exactly. -- `agents/openai.yaml` exists and still matches the skill's current purpose when bundle metadata was part of the task. -- the skill is repository-owned and still the smallest credible answer to the problem. -- the top-level `## Referenced skills` section exists immediately after the H1 for any created or materially revised `SKILL.md`. -- every non-`None` referenced-skill entry maps to a real repository skill or an explicitly justified external/on-demand skill. -- the referenced-skill index matches all skill names used later for loading, routing, comparison, or delegation. -- referenced-skill wording agrees with nearby skills that cite or are cited by the touched skill, and no entry implies eager loading unless the active contract explicitly requires it. -- the description matches the real trigger without describing the workflow. -- the description is strong enough that repository-owned skill requests should retrieve this skill before the generic OpenAI one. -- the result makes rejection, reuse, and in-place tightening as natural as creation or replacement. -- the skill still points to the right adjacent owner when the work is actually catalog governance or agent authoring. -- the skill reads like a reusable guide instead of a one-off narrative. -- the edited skill does not repeat agent-owned routing or boundary language. -- the edited skill points to reference-owned deep material instead of copying it back into `SKILL.md`. -- the edited `SKILL.md` does not point at another skill's internal files. -- the touched bundle is self-contained by default: required instructions, - examples, fixtures, metadata, and deterministic automation are bundle-local, - or any exception is explicit and justified by the contract. -- self-references inside the touched bundle use bundle-relative paths unless a - different path shape is required by the contract. -- any paired agent or local references still agree with the skill boundary when they exist. -- when direct-copy portability or out-of-repo execution is part of the skill contract, the bundle still contains the required runnable automation and repository scripts do not become the only operating engine. -- OpenAI-side scaffolding or validation was invoked only when the remaining work actually required it. -- the retrieval and pressure tests appropriate to the skill type have actually been run. -- the body did not become a maintenance fork of generic OpenAI bundle documentation. - -## Repository follow-up - -- Update nearby routing or support-skill references when this skill changes the visible local entrypoint. -- Re-check `.github/INVENTORY.md` whenever a repository-owned skill is added, retired, renamed, or replaced. -- Escalate to `local-agent-sync-external-resources` through `local-sync-external-resources` when the change becomes catalog governance instead of one-skill authoring. - -## Common mistakes - -- Writing a skill because a request feels familiar, not because the repository needs a reusable owner. -- Treating `openai-skill-creator` as irrelevant when it should be the bundle-design core. -- Mixing catalog governance into a repo-owned skill authoring contract. -- Turning every change into a new skill instead of tightening or reusing an existing one. -- Using a long description that tells the agent what to do instead of when to load the skill. -- Skipping `agents/openai.yaml` even though the repository expects it for internal skills. -- Skipping the baseline and rationalizing the change as "small enough". -- Copying OpenAI bundle anatomy or bundle-process weight wholesale into the local wrapper instead of selecting only what improves the repository-owned owner. +Completion criterion: structural validation passes, routing fallout is resolved, and +before/after measurements are recorded. diff --git a/.github/skills/internal-skill-creator/agents/openai.yaml b/.github/skills/internal-skill-creator/agents/openai.yaml index 66b911c0..96322f85 100644 --- a/.github/skills/internal-skill-creator/agents/openai.yaml +++ b/.github/skills/internal-skill-creator/agents/openai.yaml @@ -1,4 +1,4 @@ interface: display_name: "Internal Skill Creator" - short_description: "First entrypoint for repo skill work" - default_prompt: "Use $internal-skill-creator first to create or revise a repository-owned skill in .github/skills/. Establish the local boundary, keep the touched bundle self-contained with bundle-local references and automation by default, then load the embedded OpenAI skill-creation workflow only for the remaining generic bundle work." + short_description: "Create and revise repository-owned skills" + default_prompt: "Use $internal-skill-creator to create or revise a repository-owned skill. Apply /mattpocock-writing-great-skills as the core method, prefix cross-skill invocations with `/`, run proportional evaluation, then complete repository closure." diff --git a/.github/skills/internal-skill-creator/references/authoring-and-evaluation.md b/.github/skills/internal-skill-creator/references/authoring-and-evaluation.md new file mode 100644 index 00000000..d05a80ca --- /dev/null +++ b/.github/skills/internal-skill-creator/references/authoring-and-evaluation.md @@ -0,0 +1,38 @@ +# Authoring and Proportional Evaluation + +## Intent contract + +Recover known answers from the request and repository evidence. Capture the +capability, invocation conditions, expected output, constraints, dependencies, +success criteria, validation path, and anti-scope. + +## Evaluation selection + +Classify each candidate branch as applicable, skipped, or blocked. Objective +file transformations and fixed workflows use executable checks. Subjective +writing, design, and judgment work use human review. + +## Baselines + +For a material revision, compare with the previous version. For a new skill, +compare with the same task without it when isolation is available. Otherwise, +record the gap and use the closest focused validator plus human review. + +## Evidence and human review + +For each applicable branch, record the prompt or fixture, expected and observed +behavior, review method, and status. Present subjective outputs to the user +before changing them from agent judgment alone. Generalize feedback; do not +optimize only for sampled prompts. + +## Description trigger checks + +Test realistic should-trigger and near-miss prompts. Include the main branch +and a competing-owner case. When tuning against enough cases, reserve a holdout +set. Record a gap when the runtime cannot measure invocation. + +## Iteration stop conditions + +Stop when the accepted prompts and validators pass, the user accepts subjective +outputs, or another iteration adds no decision-relevant evidence. Keep required +evidence gaps blocked. diff --git a/.github/skills/internal-skill-creator/references/writing-skills-checklist.md b/.github/skills/internal-skill-creator/references/writing-skills-checklist.md deleted file mode 100644 index a35831bc..00000000 --- a/.github/skills/internal-skill-creator/references/writing-skills-checklist.md +++ /dev/null @@ -1,105 +0,0 @@ -# Writing Skills Checklist - -Load this reference when creating a new repository-owned skill or materially revising an existing one. - -This is a local distilled checklist informed by external skill-authoring guidance. It exists to keep the wrapper self-contained without cloning the upstream bundle. - -## Core posture - -- Treat a skill as a reusable guide for future agents, not as a narrative about one past task. -- Prefer executable process over prose: steps, evidence, and exit criteria must change what a future agent does. -- Iron law: do not create or materially revise a skill without first observing the failure, miss, or ambiguity it must fix. -- Use the same proof standard for edits as for new skills. -- Check frontmatter validity before reviewing route quality or body wording. Structural breakage outranks content cleanup. -- Compare the target skill against the closest neighboring owners before deciding the fix. Route quality is a lane-level property. -- Keep `## Referenced skills` immediately after the H1 for any created or materially revised `SKILL.md`. - -## Generic skill shape - -- Required sections are conditional on behavior, not a rigid template. Frontmatter, H1, trigger-focused `description:`, a clear operating contract, and validation guidance are the stable minimum for material repository-owned skills. -- Conditional sections such as `## Referenced skills`, `## When to use`, `## When not to use`, local references, scripts, examples, and fixtures should appear only when they improve routing, boundary clarity, portability, or maintenance. -- Do not require every section for every skill. A small single-owner skill should stay small when extra sections would only repeat the trigger or body. -- Treat `## Referenced skills` as an audit index, not a preload list. Do not load referenced skills from this section alone; load another skill only when the active task, file, framework, runtime, blocker, validation path, or explicit user request proves that owner is needed. -- Remove duplicated responsibility, not useful trigger reinforcement. Keep short repeated safety or retrieval language when it protects activation, stop conditions, or claim discipline. -- Preserve a working `description:` during cleanup unless the observed baseline failure is routing itself. - -## Discovery and retrieval - -- Keep `description:` focused on when the skill should load. -- Avoid describing the workflow in `description:`. That creates shortcuts and weakens body retrieval. -- Make `description:` read like realistic user intent, not like a capability summary or mini playbook. -- If `description:` names too many adjacent lanes, treat it as overlap until proven otherwise. -- Add file extensions or path tokens in `description:` only when they materially disambiguate the owner. Do not add long suffix lists when path-based routing or the skill body already proves the lane. -- Use words an agent would actually search for: symptoms, overlaps, file paths, task verbs, and validation terms. -- Prefer direct action-shaped names when naming a new skill. - -## Tightening strategy - -- Prefer the smallest fix that solves the miss: one description tighten, one boundary note, or one misleading phrase removed. -- Do not rewrite a long body just because it is long. Rewrite only when it duplicates the route, duplicates another owner, or keeps reference material inline. -- A clean "tighten" outcome is often better than expanding the skill. - -## Core-backed wrappers - -- Compare the wrapper against its core before editing. Inventory which responsibilities the core already owns and remove local restatements instead of polishing them. -- Keep the wrapper limited to its retrieval trigger, repository-local policy, and proven environment fallbacks that the core cannot know. -- Do not restate the core's workflow, decision logic, output contract, or validation procedure. Reference the core by skill name and owner behavior. -- Check the wrapper `SKILL.md`, `agents/openai.yaml`, paired agent, focused tests, and nearby routing text for conflicting or stale contracts. -- Verify external assumptions such as required paths, setup commands, integrations, and runtime capabilities against the current repository before adding a local fallback. -- Structural validation is not semantic alignment. Compare the final wrapper and core responsibilities explicitly, then search for removed owners, stale workflow terms, and conflicting output rules. -- Measure token change with the same method before and after cleanup, but preserve a working trigger and required local safeguards even when they cost a small number of tokens. - -## Token discipline - -- Keep `SKILL.md` lean and move deeper material into `references/` or reusable tools only when justified. -- If the skill sits behind a paired agent, keep route and boundary language out of `SKILL.md`. -- Preserve a working `description:` during token cuts unless the baseline shows routing is wrong. -- When touching a skill bundle, verify that the bundle stays self-contained by - default: required instructions, examples, fixtures, metadata, and - deterministic automation should live inside the bundle unless the contract - explicitly documents an exception. -- For material skill-bundle revisions, measure the touched bundle or loaded files before and after the first material patch with the same token estimate; if no exact script exists, state the closest measurement. -- Prefer focused references or reusable scripts over broad context dumps when a change needs supporting detail. -- Cross-reference reusable material instead of restating it. -- For lightweight or umbrella skills, treat `## Referenced skills` as an audit index, not a preload bundle. State that named skills stay on-demand and load only when the file, framework, runtime, blocker, or validation path proves that owner. -- If local references own the deep tables, templates, or examples, point to them instead of copying them back into `SKILL.md`. -- Prefer moving static lookup tables, starter templates, and detailed taxonomies into `references/`. -- Prefer new `references/` over new `scripts/` unless the workflow is deterministic, repeated, and execution-heavy. -- If the skill must stay direct-copy portable or runnable outside the source repository, keep the required deterministic automation, loaders, and dependency bootstrap inside the bundle; repository scripts may wrap that engine but should not be the only runnable owner. -- Prefer bundle-relative self-references to files under `references/`, - `scripts/`, and `fixtures/` over repository-rooted paths to the same bundle - files. -- Reference other skills by skill name and owner behavior, not by file paths inside their bundles. -- Keep `## Referenced skills` as an audit index: one backticked skill name plus one short owner-behavior phrase per item, or `- None.` when no other skill is referenced. -- Update the referenced-skill index whenever the body adds, removes, renames, routes to, delegates to, or compares against another skill. -- When a referenced-skill rule changes, compare the touched skill with nearby skills that cite it or that it cites. Keep optional owners lazy and on-demand unless the active contract explicitly requires loading. -- Prefer one strong example over several repetitive ones. - -## Test posture - -- Run a baseline scenario without the skill or before the edit and capture the failure. -- Re-run the same scenario with the skill after the change. -- Re-run at least one neighboring-owner scenario when the edit changes routing or boundary wording. -- Add a misuse, pressure, or counterexample test based on the skill type. -- Make every verification item evidence-shaped: command output, diff evidence, rendered artifact, or an explicit validation gap. -- When the bundle owns runnable automation, validate the bundled entrypoint directly and validate any repository wrapper separately. -- Verify that every listed referenced skill exists or is explicitly marked as external/on-demand, and that no later skill reference is missing from the index. -- Verify that related skills use compatible referenced-skill wording and do not turn optional owners into preload instructions. - -## Skill-type checks - -- Discipline: test pressure, loopholes, and rationalizations. -- Technique: test failure case, success case, and one misuse case. -- Pattern: test recognition, correct use, and counterexample boundaries. -- Reference: test retrieval, correct application, and common gaps. - -## Red flags - -- "This is obvious." -- "It is only wording." -- "We can test later." -- "This should probably become a new skill." -- "I already know what the skill should say." -- The section explains a topic but does not alter trigger choice, workflow steps, evidence, or exit criteria. - -When one of these appears, stop and re-check the baseline and boundary. diff --git a/.github/skills/local-agent-sync-external-resources/SKILL.md b/.github/skills/local-agent-sync-external-resources/SKILL.md index 6a5d6483..eb049272 100644 --- a/.github/skills/local-agent-sync-external-resources/SKILL.md +++ b/.github/skills/local-agent-sync-external-resources/SKILL.md @@ -9,10 +9,15 @@ description: Audit, plan, or apply declared external resource refreshes through This skill owns the manifest-driven CLI that stages, validates, and applies declared external resource refreshes safely. The single public entrypoint is -`scripts/sync_external_resources.py`. +`scripts/sync_external_resources.py`. Bundle siblings: `references/`, +`patches/`, `agents/openai.yaml`, `scripts/`. ## Modes +- `prepare`: the only mode that uses the network. Fetches pinned Git content + into a repository-keyed partial-clone cache and exports only manifest-declared + asset paths into verified snapshots under `sources/`. All other modes are + offline and consume these verified snapshots. - `audit`: parse all registries, validate local paths, canonical names, hashes, watchlist shape, and dirty state. Does not fetch or write. - `plan`: require an external workspace, build and validate the complete @@ -21,12 +26,70 @@ declared external resource refreshes safely. The single public entrypoint is generate one patch, run `git apply --check`, apply once, rebuild inventory, and rerun scoped validation. +## Pinned Content Only + +- The manifest full commit SHA is the sole accepted source identity. +- Only manifest-declared `upstream` paths are materialized. +- No tags, no submodules, no local branches, no remote-tracking branches. +- No `git pull`, no argumentless `git fetch`, no `git remote update`. +- No package managers: `pip`, `uv`, `npm`, `brew`, `yarn`, `pnpm` are forbidden. + +## Managed Skill Reference Normalization + +- A source may set `rewrite_skill_references: true` to rewrite slash commands and + backtick skill references from each declared upstream asset basename to its + declared `canonical_name` during candidate normalization. +- Use `skill_reference_aliases` for upstream command names that do not match an + asset basename. Keep aliases source-local and point them only at declared + canonical names. +- The `mattpocock-skills` source uses the `mattpocock-` canonical prefix for + every imported skill. The upstream `/grilling` reference is normalized to + the repository-owned `grill-me` through a declared source replacement, + not through a source-local alias. +- References to undeclared skills remain unchanged and must be reported as + unresolved dependencies; do not silently import or rewrite them. + +## Guided Question Contract + +- Candidate normalization appends a repository-owned contract to + `superpowers-brainstorming` and `grill-me` whenever either canonical skill is + managed, regardless of its source or upstream wording. +- The contract requires numbered bulk question blocks. Every question includes + a brief `Recommendation`, `Why`, and `Default if accepted`. +- The appended contract explicitly overrides upstream one-question-at-a-time + pacing. A single remaining blocker is still a numbered one-item block. +- The append is marker-based and idempotent. Do not replace it with a + context-sensitive replay patch. + ## Workspace Convention -- Use a repo-local external workspace under `tmp/sync-externals-skills/`. -- Keep prepared source checkouts under the workspace `sources/` directory, +- The runtime workspace must be outside this repository. +- Use an external workspace such as `../cloud-strategy.github-external-refresh`. +- Keep prepared source snapshots under the workspace `sources/` directory, unless an explicit `--source-root` is supplied. -- Do not use `/private/tmp` in the canonical examples or guidance. +- The Git object cache lives under `/cache/repositories/`. + +## Prepare Cold and Warm Flow + +- Cold `prepare` fetches each declared source SHA into the partial-clone cache, + verifies the commit object, and exports only declared upstream paths into + atomic snapshots under `/sources//`. +- Warm `prepare` finds the pin ref already cached and performs no fetch, + reporting `cached` status and zero added cache bytes. +- `--rebuild-cache` forces a fresh cache rebuild beside the active one, + replacing it only after verification. The rebuild reports `cache_status` + `rebuilt` only after the fresh cache replaces the active one. + +## TSV Output + +- `--format tsv` emits escaped, deterministic, lexically sorted records. +- Header: `record\tkey\tstatus\tvalue`. +- Record types: `summary`, `source`, `metric`, `change`, `override`, + `validation`, `blocker`. +- `metric` and per-source `validation` rows use `key` `.`, + `status` `ok` or `fail`, and the measured value in `value`. +- `--format text` remains the default for operators. +- `--format json` retains backward-compatible keys. ## Safety @@ -39,27 +102,32 @@ declared external resource refreshes safely. The single public entrypoint is ## Workflow -1. Run `audit` to validate the manifest, overrides, and local dirty-state without blocking on a dirty managed target. -2. Prepare source checkouts explicitly under the chosen source root. -3. Run `plan` with `--workspace` and, when needed, `--source-root` to confirm the candidate can be built. +1. Run `prepare` to fetch pinned content into verified snapshots. +2. Run `audit` to validate the manifest, overrides, and local dirty-state. +3. Run `plan` with `--workspace` to confirm the candidate can be built. 4. Review the changed-path summary and override replay results. -5. Run `apply` only after `plan` succeeds; `apply` does not fetch sources implicitly. - -If `plan` or `apply` reports `Missing upstream paths`, the message will name the -expected root such as `tmp/sync-externals-skills//sources`. -Populate that root first or pass an explicit `--source-root`. +5. Run `apply` only after `plan` succeeds; `apply` does not fetch sources. -Prefer reusing an existing prepared workspace under `tmp/sync-externals-skills/` -before creating a new one. +If `plan` or `apply` reports `Missing prepared source metadata`, run `prepare` +first. ## Canonical Commands ```bash -python3 .github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py audit -python3 .github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py plan --workspace tmp/sync-externals-skills/cloud-strategy-github-external-refresh --source-root tmp/sync-externals-skills/cloud-strategy-github-external-refresh/sources -python3 .github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py apply --workspace tmp/sync-externals-skills/cloud-strategy-github-external-refresh --source-root tmp/sync-externals-skills/cloud-strategy-github-external-refresh/sources +python3 .github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py prepare --workspace ../cloud-strategy.github-external-refresh --format tsv +python3 .github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py audit --format tsv +python3 .github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py plan --workspace ../cloud-strategy.github-external-refresh --format tsv +python3 .github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py apply --workspace ../cloud-strategy.github-external-refresh --format tsv ``` +## Live Network Benchmark (Separate Authorization Required) + +Run `prepare` (see `## Canonical Commands`) against the two largest sources +(`github-awesome-copilot` and `sickn33-antigravity`) with 242 MiB and 318 MiB +baselines. Require at least 90% reduction in `materialized_bytes`. Run again +and require `cached` with zero added cache bytes. The command must not alter +manifest refs or repository targets. Do not run without separate authorization. + ## Override Rules - Every approved imported in-place override must be registered in @@ -80,7 +148,7 @@ python3 .github/skills/local-agent-sync-external-resources/scripts/sync_external ## Output Report: mode, workspace, source root when used, managed count, changed paths, -override results, validation, and blockers. +override results, source metrics, validation, and blockers. ## Anti-Scope @@ -88,3 +156,4 @@ override results, validation, and blockers. - Do not add a plugin system, concurrency, compatibility aliases, legacy fallback paths, or a generic sync framework. - Do not perform a live network refresh unless the user separately authorizes it. +- Do not use package managers or mutable branch updates in any sync mode. diff --git a/.github/skills/local-agent-sync-external-resources/patches/grill-me.patch b/.github/skills/local-agent-sync-external-resources/patches/grill-me.patch deleted file mode 100644 index ecb41998..00000000 --- a/.github/skills/local-agent-sync-external-resources/patches/grill-me.patch +++ /dev/null @@ -1,45 +0,0 @@ -diff --git a/.github/skills/grill-me/SKILL.md b/.github/skills/grill-me/SKILL.md -index 4b857d0..ff05fa6 100644 ---- a/.github/skills/grill-me/SKILL.md -+++ b/.github/skills/grill-me/SKILL.md -@@ -1,7 +1,37 @@ - --- - name: grill-me --description: A relentless interview to sharpen a plan or design, which also creates docs (ADR's and glossary) as we go. --disable-model-invocation: true -+description: A relentless interview to sharpen a plan or design. - --- - --Run a `/grilling` session, using the `/domain-modeling` skill. -+# Grill Me -+ -+## Referenced skills -+ -+- None. -+ -+## Interview Approach -+ -+Interview me relentlessly about every aspect of this plan, design, or action -+context until we reach a shared understanding. Walk down each branch of the -+decision tree, resolving dependencies between decisions one-by-one. -+ -+Before asking, inspect the repository, codebase, documentation, or local files for answers that can be recovered from evidence. -+ -+By default, ask the full initial question set in one numbered list. -+ -+Structure the list by decision branch and dependency order. Start with goal and scope, then constraints, architecture or options, risks and failure modes, rollout, and validation as relevant. -+ -+For each numbered question, use this format: Question, Recommendation, Why, and Default if accepted. Make the recommendation detailed enough to explain what you want to decide, why it matters, and what answer you would choose by default. -+ -+Treat your recommendations as accepted unless the user says otherwise. The user may override any recommendation by referencing the question number or giving different direction. -+ -+Do not treat accepted recommendations as the end of the grilling process. If accepted defaults create contradictions, weak assumptions, unresolved risks, or dependent decisions, surface them explicitly. -+ -+After the initial numbered list, ask one question at a time only for unresolved ambiguity, dependent follow-up decisions, or branches that cannot be settled from the user's bulk response. -+ -+A caller may override the follow-up pacing with iterative numbered blocks. When a caller declares that override, replace the default one-at-a-time follow-up with the caller's pacing. -+ -+Do not ask questions that can be answered by exploring the codebase, documentation, or local files. -+ -+End by summarizing the resolved decisions, explicit assumptions, and any -+unresolved questions the user chose to accept or defer. diff --git a/.github/skills/local-agent-sync-external-resources/patches/mattpocock-handoff-tmp-path.patch b/.github/skills/local-agent-sync-external-resources/patches/mattpocock-handoff-tmp-path.patch index 6a1db043..110c82a1 100644 --- a/.github/skills/local-agent-sync-external-resources/patches/mattpocock-handoff-tmp-path.patch +++ b/.github/skills/local-agent-sync-external-resources/patches/mattpocock-handoff-tmp-path.patch @@ -2,11 +2,17 @@ diff --git a/.github/skills/mattpocock-handoff/SKILL.md b/.github/skills/mattpoc index 37e6be0..0000000 100644 --- a/.github/skills/mattpocock-handoff/SKILL.md +++ b/.github/skills/mattpocock-handoff/SKILL.md -@@ -5,6 +5,6 @@ argument-hint: "What will the next session be used for?" +@@ -5,11 +5,11 @@ disable-model-invocation: true --- - + -Write a handoff document summarising the current conversation so a fresh agent can continue the work. Save to the temporary directory of the user's OS - not the current workspace. +Write a handoff document summarising the current conversation so a fresh agent can continue the work. Save it under `tmp/handoff/` in the current workspace, creating that directory if needed. - + Include a "suggested skills" section in the document, which suggests skills that the agent should invoke. + +-Do not duplicate content already captured in other artifacts (specs, plans, ADRs, issues, commits, diffs). Reference them by path or URL instead. ++Do not duplicate content already captured in other artifacts (PRDs, plans, ADRs, issues, commits, diffs). Reference them by path or URL instead. + + Redact any sensitive information, such as API keys, passwords, or personally identifiable information. + diff --git a/.github/skills/local-agent-sync-external-resources/patches/mattpocock-improve-codebase-architecture-delegated-invocation.patch b/.github/skills/local-agent-sync-external-resources/patches/mattpocock-improve-codebase-architecture-delegated-invocation.patch new file mode 100644 index 00000000..dff18eb3 --- /dev/null +++ b/.github/skills/local-agent-sync-external-resources/patches/mattpocock-improve-codebase-architecture-delegated-invocation.patch @@ -0,0 +1,12 @@ +diff --git a/.github/skills/mattpocock-improve-codebase-architecture/SKILL.md b/.github/skills/mattpocock-improve-codebase-architecture/SKILL.md +index 0000000..0000000 100644 +--- a/.github/skills/mattpocock-improve-codebase-architecture/SKILL.md ++++ b/.github/skills/mattpocock-improve-codebase-architecture/SKILL.md +@@ -1,7 +1,6 @@ + --- + name: mattpocock-improve-codebase-architecture + description: Scan a codebase for deepening opportunities, present them as a visual HTML report, then grill through whichever one you pick. +-disable-model-invocation: true + --- + + # Improve Codebase Architecture diff --git a/.github/skills/local-agent-sync-external-resources/patches/mattpocock-writing-great-skills-delegated-invocation.patch b/.github/skills/local-agent-sync-external-resources/patches/mattpocock-writing-great-skills-delegated-invocation.patch new file mode 100644 index 00000000..ed71effc --- /dev/null +++ b/.github/skills/local-agent-sync-external-resources/patches/mattpocock-writing-great-skills-delegated-invocation.patch @@ -0,0 +1,12 @@ +diff --git a/.github/skills/mattpocock-writing-great-skills/SKILL.md b/.github/skills/mattpocock-writing-great-skills/SKILL.md +index 0000000..0000000 100644 +--- a/.github/skills/mattpocock-writing-great-skills/SKILL.md ++++ b/.github/skills/mattpocock-writing-great-skills/SKILL.md +@@ -1,6 +1,5 @@ + --- + name: mattpocock-writing-great-skills +-description: Reference for writing and editing skills well — the vocabulary and principles that make a skill predictable. +-disable-model-invocation: true ++description: Predictability review and revision stage applied after authoring. Loaded by internal-skill-creator after the Anthropic authoring stage when delegating skill work. + --- + diff --git a/.github/skills/local-agent-sync-external-resources/patches/openai-docx-render-docx.patch b/.github/skills/local-agent-sync-external-resources/patches/openai-docx-render-docx.patch deleted file mode 100644 index d5838cee..00000000 --- a/.github/skills/local-agent-sync-external-resources/patches/openai-docx-render-docx.patch +++ /dev/null @@ -1,9 +0,0 @@ -diff --git a/.github/skills/openai-docx/scripts/render_docx.py b/.github/skills/openai-docx/scripts/render_docx.py -index 907ec89..be4d0d3 100644 ---- a/.github/skills/openai-docx/scripts/render_docx.py -+++ b/.github/skills/openai-docx/scripts/render_docx.py -@@ -1,3 +1,4 @@ -+#!/usr/bin/env python3 - import argparse - import os - import re diff --git a/.github/skills/local-agent-sync-external-resources/patches/openai-spreadsheet.patch b/.github/skills/local-agent-sync-external-resources/patches/openai-spreadsheet.patch deleted file mode 100644 index 3aea6128..00000000 --- a/.github/skills/local-agent-sync-external-resources/patches/openai-spreadsheet.patch +++ /dev/null @@ -1,31 +0,0 @@ -diff --git a/.github/skills/openai-spreadsheet/SKILL.md b/.github/skills/openai-spreadsheet/SKILL.md -index 4f41e8d..2236908 100644 ---- a/.github/skills/openai-spreadsheet/SKILL.md -+++ b/.github/skills/openai-spreadsheet/SKILL.md -@@ -1,5 +1,5 @@ - --- --name: openai-spreadsheet -+name: "openai-spreadsheet" - description: "Use when tasks involve creating, editing, analyzing, or formatting spreadsheets (`.xlsx`, `.csv`, `.tsv`) with formula-aware workflows, cached recalculation, and visual review." - --- - -@@ -33,6 +33,19 @@ IMPORTANT: System and user instructions always take precedence. - - Use `openpyxl.chart` for native Excel charts when needed. - - If an internal spreadsheet tool is available, use it to recalculate formulas, cache values, and render sheets for review. - -+## Structured Data Evidence Budget -+- For large `.xlsx`, `.csv`, and `.tsv` work, keep user-facing evidence compact: -+ report schema or headers, row counts, column counts, targeted anomalies, -+ checksums or hashes when useful, sampled examples, and validation gaps. -+- Start discovery with headers plus a small sample, then move to deterministic -+ full-file checks when correctness depends on the whole dataset. -+- Sampling does not replace full-file validation for transforms, merges, -+ source-link checks, empty-row checks, column moves, stable ID generation, -+ duplicate ID detection, or reconciliation. -+- Preserve source links, formulas, formatting where applicable, empty-row -+ anomalies, duplicate IDs, stable generated IDs, and missing column data as -+ material integrity checks. -+ - ## Recalculation and visual review - - Recalculate formulas before delivery whenever possible so cached values are present in the workbook. - - Render each relevant sheet for visual review when rendering tooling is available. diff --git a/.github/skills/local-agent-sync-external-resources/patches/superpowers-brainstorming.patch b/.github/skills/local-agent-sync-external-resources/patches/superpowers-brainstorming.patch deleted file mode 100644 index 29768f88..00000000 --- a/.github/skills/local-agent-sync-external-resources/patches/superpowers-brainstorming.patch +++ /dev/null @@ -1,146 +0,0 @@ -diff --git a/.github/skills/superpowers-brainstorming/SKILL.md b/.github/skills/superpowers-brainstorming/SKILL.md -index ddfb2c4..47afc6e 100644 ---- a/.github/skills/superpowers-brainstorming/SKILL.md -+++ b/.github/skills/superpowers-brainstorming/SKILL.md -@@ -7,7 +7,7 @@ description: "You MUST use this before any creative work - creating features, bu - - Help turn ideas into fully formed designs and specs through natural collaborative dialogue. - --Start by understanding the current project context, then ask questions one at a time to refine the idea. Once you understand what you're building, present the design and get user approval. -+Start by understanding the current project context, then Ask clarifying questions in numbered bulk question blocks. Each question must include a short recommendation, a short reason for that recommendation, and the default that will be treated as accepted when the user accepts the suggestions. Once you understand what you're building, decide whether a retained spec adds real design value or whether moving directly to an implementation plan is the better next step. - - - Do NOT invoke any implementation skill, write any code, scaffold any project, or take any implementation action until you have presented a design and the user has approved it. This applies to EVERY project regardless of perceived simplicity. -@@ -23,34 +23,38 @@ You MUST create a task for each of these items and complete them in order: - - 1. **Explore project context** — check files, docs, recent commits - 2. **Offer the visual companion just-in-time** — NOT upfront. The first time a question would genuinely be clearer shown than described, offer it then (its own message); on approval its browser tab opens for you. If no visual question ever arises, never offer it. See the Visual Companion section below. --3. **Ask clarifying questions** — one at a time, understand purpose/constraints/success criteria -+3. **Ask clarifying questions** — use numbered bulk question blocks with `Question`, `Recommendation`, `Why`, and `Default if accepted` - 4. **Propose 2-3 approaches** — with trade-offs and your recommendation - 5. **Present design** — in sections scaled to their complexity, get user approval after each section --6. **Write design doc** — save to `tmp/superpowers/specs/YYYY-MM-DD--design.md` and commit --7. **Spec self-review** — quick inline check for placeholders, contradictions, ambiguity, scope (see below) --8. **User reviews written spec** — ask user to review the spec file before proceeding --9. **Transition to implementation** — invoke writing-plans skill to create implementation plan -+6. **Run the Design-Depth Gate** — choose `Decision: direct plan` or `Decision: spec first`, and tell the user why -+7. **Write design doc when needed** — if the gate chooses `spec first` and the user wants a retained file, save it to `tmp/superpowers/specs/YYYY-MM-DD--design.md`; never commit files from `tmp/` -+8. **Spec self-review when needed** — quick inline check for placeholders, contradictions, ambiguity, scope (see below) -+9. **User reviews written spec when needed** — ask user to review the spec file before proceeding -+10. **Transition to implementation planning** — invoke writing-plans skill after the user approves either the design or the direct-plan recommendation - - ## Process Flow - - ```dot - digraph brainstorming { - "Explore project context" [shape=box]; -- "Ask clarifying questions" [shape=box]; -+ "Ask bulk clarifying questions" [shape=box]; - "Propose 2-3 approaches" [shape=box]; - "Present design sections" [shape=box]; - "User approves design?" [shape=diamond]; -+ "Design-Depth Gate" [shape=diamond]; - "Write design doc" [shape=box]; - "Spec self-review\n(fix inline)" [shape=box]; - "User reviews spec?" [shape=diamond]; - "Invoke writing-plans skill" [shape=doublecircle]; - -- "Explore project context" -> "Ask clarifying questions"; -- "Ask clarifying questions" -> "Propose 2-3 approaches"; -+ "Explore project context" -> "Ask bulk clarifying questions"; -+ "Ask bulk clarifying questions" -> "Propose 2-3 approaches"; - "Propose 2-3 approaches" -> "Present design sections"; - "Present design sections" -> "User approves design?"; - "User approves design?" -> "Present design sections" [label="no, revise"]; -- "User approves design?" -> "Write design doc" [label="yes"]; -+ "User approves design?" -> "Design-Depth Gate" [label="yes"]; -+ "Design-Depth Gate" -> "Invoke writing-plans skill" [label="direct plan"]; -+ "Design-Depth Gate" -> "Write design doc" [label="spec first"]; - "Write design doc" -> "Spec self-review\n(fix inline)"; - "Spec self-review\n(fix inline)" -> "User reviews spec?"; - "User reviews spec?" -> "Write design doc" [label="changes requested"]; -@@ -58,7 +62,7 @@ digraph brainstorming { - } - ``` - --**The terminal state is invoking writing-plans.** Do NOT invoke frontend-design, mcp-builder, or any other implementation skill. The ONLY skill you invoke after brainstorming is writing-plans. -+**The terminal state is invoking writing-plans.** Do NOT invoke frontend-design, mcp-builder, or any other implementation skill. The ONLY skill you invoke after brainstorming is writing-plans. The path to writing-plans may be either `Decision: direct plan` or `Decision: spec first`. - - ## The Process - -@@ -67,9 +71,13 @@ digraph brainstorming { - - Check out the current project state first (files, docs, recent commits) - - Before asking detailed questions, assess scope: if the request describes multiple independent subsystems (e.g., "build a platform with chat, file storage, billing, and analytics"), flag this immediately. Don't spend questions refining details of a project that needs to be decomposed first. - - If the project is too large for a single spec, help the user decompose into sub-projects: what are the independent pieces, how do they relate, what order should they be built? Then brainstorm the first sub-project through the normal design flow. Each sub-project gets its own spec → plan → implementation cycle. --- For appropriately-scoped projects, ask questions one at a time to refine the idea -+- For appropriately-scoped projects, ask clarifying questions in numbered bulk question blocks to refine the idea - - Prefer multiple choice questions when possible, but open-ended is fine too --- Only one question per message - if a topic needs more exploration, break it into multiple questions -+- Each numbered question must use this format: `Question`, `Recommendation`, `Why`, and `Default if accepted` -+- The recommendation and why must be clear and brief -+- The user may accept all suggested defaults, accept only some numbered defaults, or override any numbered recommendation -+- Accepted defaults do not mean discovery is complete. If the accepted answers create contradictions, weak assumptions, unresolved risks, or dependent decisions, ask another focused numbered bulk question block. -+- Ask another focused numbered bulk question block only for unresolved, dependent, or reopened branches. Do not ask questions that project evidence can answer. - - Focus on understanding: purpose, constraints, success criteria - - **Exploring approaches:** -@@ -86,6 +94,23 @@ digraph brainstorming { - - Cover: architecture, components, data flow, error handling, testing - - Be ready to go back and clarify if something doesn't make sense - -+**Design-Depth Gate:** -+ -+Before writing a retained spec, decide whether the spec has meaningful marginal value over going directly to an implementation plan. -+ -+Choose `Decision: direct plan` when the target, owner, scope, constraints, rejected alternatives, and validation path are already clear, and a retained spec would mostly duplicate the implementation plan. -+ -+Choose `Decision: spec first` when product, design, architecture, data flow, user experience, rollout, or risk decisions are still material enough that a retained spec would reduce the chance of building the wrong thing. -+ -+In both cases, tell the user the decision and one short reason: -+ -+- `Decision: direct plan` -+- `Why: ` -+- `Decision: spec first` -+- `Why: ` -+ -+Ask for user approval before invoking writing-plans. Direct plan skips the retained spec, not the user approval gate. -+ - **Design for isolation and clarity:** - - - Break the system into smaller units that each have one clear purpose, communicate through well-defined interfaces, and can be understood and tested independently -@@ -103,10 +128,11 @@ digraph brainstorming { - - **Documentation:** - --- Write the validated design (spec) to `tmp/superpowers/specs/YYYY-MM-DD--design.md` -+- Write the validated design (spec) to chat by default -+- Persist it to `tmp/superpowers/specs/YYYY-MM-DD--design.md` only when `Decision: spec first` is chosen and the user explicitly wants a retained file - - (User preferences for spec location override this default) - - Use elements-of-style:writing-clearly-and-concisely skill if available --- Commit the design document to git -+- Never commit files from `tmp/` - - **Spec Self-Review:** - After writing the spec document, look at it with fresh eyes: -@@ -121,18 +147,18 @@ Fix any issues inline. No need to re-review — just fix and move on. - **User Review Gate:** - After the spec review loop passes, ask the user to review the written spec before proceeding: - --> "Spec written and committed to ``. Please review it and let me know if you want to make any changes before we start writing out the implementation plan." -+> "Spec written to ``. Please review it and let me know if you want to make any changes before we start writing out the implementation plan." - - Wait for the user's response. If they request changes, make them and re-run the spec review loop. Only proceed once the user approves. - - **Implementation:** - --- Invoke the writing-plans skill to create a detailed implementation plan -+- Invoke the writing-plans skill to create a detailed implementation plan after the user approves either `Decision: direct plan` or the reviewed spec - - Do NOT invoke any other skill. writing-plans is the next step. - - ## Key Principles - --- **One question at a time** - Don't overwhelm with multiple questions -+- **Bulk guided question blocks** - Ask the full known question set together, with recommendations, reasons, and defaults - - **Multiple choice preferred** - Easier to answer than open-ended when possible - - **YAGNI ruthlessly** - Remove unnecessary features from all designs - - **Explore alternatives** - Always propose 2-3 approaches before settling diff --git a/.github/skills/local-agent-sync-external-resources/references/imported-asset-overrides.yaml b/.github/skills/local-agent-sync-external-resources/references/imported-asset-overrides.yaml index 79d122e3..3a60e296 100644 --- a/.github/skills/local-agent-sync-external-resources/references/imported-asset-overrides.yaml +++ b/.github/skills/local-agent-sync-external-resources/references/imported-asset-overrides.yaml @@ -22,64 +22,6 @@ overrides: baseline_repo_commit: efa058a validation_note: Stop the refresh if the patch does not apply cleanly; review whether the repository-local handoff path remains the intended contract. -- id: grill-me-bulk-recommended-questions - target_path: .github/skills/grill-me/SKILL.md - source_family: mattpocock/skills - lifecycle_mode: post-refresh-patch - apply_strategy: git-apply-3way - approval: explicit-user-counter-validated - reason: Default the initial grilling pass to a dependency-ordered numbered question - set with clear recommendations that the user can accept by default, while preserving - follow-up pressure on contradictions, risks, and unresolved decisions; keep operational - gate semantics in internal-gateway-review. - patch_path: patches/grill-me.patch - expected_content_hash: 4965b2f80bed21a005e0baa241e8e9e0d502c276cdf83c394cb3feac1d4dc71b - baseline_repo_commit: efa058a - validation_note: Stop the refresh if the patch does not apply cleanly; review whether - an internal wrapper should replace the override. -- id: openai-spreadsheet-structured-data-evidence-budget - target_path: .github/skills/openai-spreadsheet/SKILL.md - source_family: openai/skills - lifecycle_mode: post-refresh-patch - apply_strategy: git-apply-3way - approval: explicit-user-counter-validated - reason: Preserve the repository-specific structured-data evidence budget so CSV, - TSV, and XLSX work stays low-token without weakening full-file integrity checks - after an imported office-skill refresh. - patch_path: patches/openai-spreadsheet.patch - expected_content_hash: f42e8ea1128c448a7ccaa42b2ca39540c5b9109a7e6e6ee4abf4e3e5e4a80722 - baseline_repo_commit: e6afb0d - validation_note: Stop the refresh if the patch does not apply cleanly; review whether - an internal wrapper should replace the override. -- id: openai-docx-executable-renderer - target_path: .github/skills/openai-docx/scripts/render_docx.py - source_family: openai/skills - lifecycle_mode: post-refresh-patch - apply_strategy: git-apply - approval: explicit-user-counter-validated - reason: Preserve a valid executable entrypoint for the bundled DOCX renderer - so repository pre-commit validation does not fail after a retained office-skill - refresh. - patch_path: patches/openai-docx-render-docx.patch - expected_content_hash: b6b1c18d5a46d81d161a57e84947ad7b2941f8095f1d52fbfe1a2c57585d7a3a - baseline_repo_commit: 45d05d7 - validation_note: Stop the refresh if the patch does not apply cleanly; review whether - the retained support-only skill should instead stop shipping an executable renderer. -- id: superpowers-brainstorming-guided-plan-gate - target_path: .github/skills/superpowers-brainstorming/SKILL.md - source_family: obra/superpowers - lifecycle_mode: post-refresh-patch - apply_strategy: git-apply-3way - approval: explicit-user-counter-validated - reason: Preserve the repository-specific brainstorming flow that asks guided numbered - bulk question blocks, keeps discovery open after accepted defaults when risks - or dependent decisions remain, and chooses between direct plan and spec first - based on the marginal value of a spec. - patch_path: patches/superpowers-brainstorming.patch - expected_content_hash: 4e44155f52c800784b3fb19be4c1e6fb8298cb49c20f611ead09b39f4998e6c7 - baseline_repo_commit: local-plan-2026-07-05 - validation_note: Stop the refresh if the patch does not apply cleanly; review whether - an internal wrapper or upstream proposal should replace the override. - id: awesome-copilot-dependabot-description-budget target_path: .github/skills/awesome-copilot-dependabot/SKILL.md source_family: github/awesome-copilot @@ -119,3 +61,30 @@ overrides: baseline_repo_commit: e986f49 validation_note: Stop the refresh if the patch does not apply cleanly; review whether an internal wrapper should replace the override. +- id: mattpocock-writing-great-skills-delegated-invocation + target_path: .github/skills/mattpocock-writing-great-skills/SKILL.md + source_family: mattpocock/skills + lifecycle_mode: post-refresh-patch + apply_strategy: git-apply + approval: explicit-user-counter-validated + reason: Narrow the imported Matt Pocock writing-great-skills description to delegation-oriented + use by internal-skill-creator and remove disable-model-invocation so the orchestrator + can invoke it programmatically during the predictability review stage. + patch_path: patches/mattpocock-writing-great-skills-delegated-invocation.patch + expected_content_hash: 41d4749ab5b4d903a8242b2e799c52ec405c35a520969bff6a43d6e788fe9254 + baseline_repo_commit: ed37663cc5fbef691ddfecd080dff42f7e7e350d + validation_note: Stop the refresh if the patch does not apply cleanly; review whether + the delegation contract still matches the intended orchestration order. +- id: mattpocock-improve-codebase-architecture-delegated-invocation + target_path: .github/skills/mattpocock-improve-codebase-architecture/SKILL.md + source_family: mattpocock/skills + lifecycle_mode: post-refresh-patch + apply_strategy: git-apply + approval: explicit-user-counter-validated + reason: Remove disable-model-invocation because internal-gateway-codebase-improvement + delegates architecture discovery to this imported method owner. + patch_path: patches/mattpocock-improve-codebase-architecture-delegated-invocation.patch + expected_content_hash: 336d2b1b9a36341d9134a7a86c041bdab8834b13e01bd106dc0e830ce59143c1 + baseline_repo_commit: ed37663cc5fbef691ddfecd080dff42f7e7e350d + validation_note: Stop the refresh if the patch does not apply cleanly; review whether + the codebase-improvement gateway still delegates to this method owner. diff --git a/.github/skills/local-agent-sync-external-resources/references/managed-resources.yaml b/.github/skills/local-agent-sync-external-resources/references/managed-resources.yaml index 444f2598..f3a8adc7 100644 --- a/.github/skills/local-agent-sync-external-resources/references/managed-resources.yaml +++ b/.github/skills/local-agent-sync-external-resources/references/managed-resources.yaml @@ -2,7 +2,7 @@ version: 1 sources: github-awesome-copilot: repository: https://github.com/github/awesome-copilot.git - ref: e986f49695491311df2774030ebe11efabd0fb77 + ref: aa280f28b1b73f9b6e6917b607eb92127b67b419 assets: - upstream: skills/agentic-eval local: .github/skills/awesome-copilot-agentic-eval @@ -36,7 +36,7 @@ sources: canonical_name: awesome-copilot-security-review obra-superpowers: repository: https://github.com/obra/superpowers.git - ref: d884ae04edebef577e82ff7c4e143debd0bbec99 + ref: 3dcbd5c4b48e02263fbf4a3c01e3fe4f81d584d9 assets: - upstream: skills/brainstorming local: .github/skills/superpowers-brainstorming @@ -79,7 +79,7 @@ sources: canonical_name: superpowers-writing-plans hashicorp-agent-skills: repository: https://github.com/hashicorp/agent-skills.git - ref: 339a113935812ad75c6ff90d418b739a021826d1 + ref: 8c6573abbd21e8094fab8f538eb5f97db63133fd assets: - upstream: terraform/code-generation/skills/terraform-search-import local: .github/skills/terraform-terraform-search-import @@ -89,20 +89,55 @@ sources: canonical_name: terraform-terraform-test mattpocock-skills: repository: https://github.com/mattpocock/skills.git - ref: efa058a349f5ce98b6115bf8b4e0d0ef9c310e0d + ref: ed37663cc5fbef691ddfecd080dff42f7e7e350d + rewrite_skill_references: true + backtick_skill_references: + - code-review + - tdd + - to-spec assets: - upstream: skills/engineering/grill-with-docs - local: .github/skills/grill-me - canonical_name: grill-me + local: .github/skills/mattpocock-grill-with-docs + canonical_name: mattpocock-grill-with-docs + - upstream: skills/engineering/domain-modeling + local: .github/skills/mattpocock-domain-modeling + canonical_name: mattpocock-domain-modeling + - upstream: skills/engineering/codebase-design + local: .github/skills/mattpocock-codebase-design + canonical_name: mattpocock-codebase-design + - upstream: skills/engineering/improve-codebase-architecture + local: .github/skills/mattpocock-improve-codebase-architecture + canonical_name: mattpocock-improve-codebase-architecture + - upstream: skills/engineering/implement + local: .github/skills/mattpocock-implement + canonical_name: mattpocock-implement + - upstream: skills/engineering/tdd + local: .github/skills/mattpocock-tdd + canonical_name: mattpocock-tdd + - upstream: skills/engineering/to-spec + local: .github/skills/mattpocock-to-spec + canonical_name: mattpocock-to-spec + - upstream: skills/engineering/setup-matt-pocock-skills + local: .github/skills/mattpocock-setup-matt-pocock-skills + canonical_name: mattpocock-setup-matt-pocock-skills + - upstream: skills/engineering/code-review + local: .github/skills/mattpocock-code-review + canonical_name: mattpocock-code-review + - upstream: skills/engineering/wayfinder + local: .github/skills/mattpocock-wayfinder + canonical_name: mattpocock-wayfinder - upstream: skills/engineering/research local: .github/skills/mattpocock-research canonical_name: mattpocock-research - upstream: skills/productivity/handoff local: .github/skills/mattpocock-handoff canonical_name: mattpocock-handoff + - upstream: skills/productivity/writing-great-skills + local: .github/skills/mattpocock-writing-great-skills + canonical_name: mattpocock-writing-great-skills vercel-labs-skills: repository: https://github.com/vercel-labs/skills.git - ref: 4ce6d48ac44c8b637db87b2102fea3baca719df1 + ref: e173b8c88f2581cfdaa1b6767c6519a08155790e assets: - upstream: skills/find-skills local: .github/skills/vercel-find-skills @@ -117,32 +152,16 @@ sources: - upstream: skills/.curated/gh-fix-ci local: .github/skills/openai-gh-fix-ci canonical_name: openai-gh-fix-ci - - upstream: skills/.system/skill-creator - local: .github/skills/openai-skill-creator - canonical_name: openai-skill-creator - - upstream: skills/.curated/pdf - local: .github/skills/openai-pdf - canonical_name: openai-pdf openai-skills-retained-doc: repository: https://github.com/openai/skills.git - ref: 45d05d75363abf13f99d09e899d61e07b8010685 - assets: - - upstream: skills/.curated/doc - local: .github/skills/openai-docx - canonical_name: openai-docx - openai-skills-retained-office: - repository: https://github.com/openai/skills.git - ref: e6afb0d74cc75d220df2faf3dd6c635c2dc6a108 + ref: 49f948faa9258a0c61caceaf225e179651397431 assets: - - upstream: skills/.curated/spreadsheet - local: .github/skills/openai-spreadsheet - canonical_name: openai-spreadsheet - - upstream: skills/.curated/slides - local: .github/skills/openai-slides - canonical_name: openai-slides + - upstream: skills/.curated/openai-docs + local: .github/skills/openai-docs + canonical_name: openai-docs sickn33-antigravity: repository: https://github.com/sickn33/antigravity-awesome-skills.git - ref: 8946c6cdc8468183426d52f85054639b3e1844ae + ref: e66fc833f2022c3534ba74af835db14c34f9a732 assets: - upstream: skills/api-design-principles local: .github/skills/antigravity-api-design-principles @@ -167,7 +186,7 @@ sources: canonical_name: antigravity-network-engineer addyosmani-agent-skills: repository: https://github.com/addyosmani/agent-skills.git - ref: 0e63d8ea962ecac51637421d8fff63702fac766b + ref: ff2df4c07e7836a092ed28e1e9b42f4d6009280c assets: - upstream: skills/code-simplification local: .github/skills/addyosmani-code-simplification @@ -175,10 +194,37 @@ sources: - upstream: skills/code-review-and-quality local: .github/skills/addyosmani-code-review-and-quality canonical_name: addyosmani-code-review-and-quality + atlassian-mcp-server: + repository: https://github.com/atlassian/atlassian-mcp-server.git + ref: f22e7075136a62baa7c10200a64884f83bf3ebe1 + assets: + - upstream: skills/search-company-knowledge + local: .github/skills/search-company-knowledge + canonical_name: search-company-knowledge + anthropic-skills: + repository: https://github.com/anthropics/skills.git + ref: b29e7cf65e5cb78a5ac33d582270551bc74a14eb + rewrite_skill_references: true + assets: + - upstream: skills/docx + local: .github/skills/anthropic-docx + canonical_name: anthropic-docx + - upstream: skills/pdf + local: .github/skills/anthropic-pdf + canonical_name: anthropic-pdf + - upstream: skills/pptx + local: .github/skills/anthropic-pptx + canonical_name: anthropic-pptx + - upstream: skills/xlsx + local: .github/skills/anthropic-xlsx + canonical_name: anthropic-xlsx normalizations: - source: obra-superpowers from: docs/superpowers to: tmp/superpowers + - source: mattpocock-skills + from: /grilling + to: /grill-me watchlist: - source_family: github/awesome-copilot upstream_id: azure-devops-pipelines.instructions.md @@ -200,10 +246,6 @@ watchlist: upstream_id: diagnose local_owner: internal-debugging reason: Root-cause diagnosis loop was extracted into a repository-owned owner. - - source_family: mattpocock/skills - upstream_id: tdd - local_owner: internal-tdd - reason: Test-first delivery rules were extracted into a repository-owned owner. - source_family: mattpocock/skills upstream_id: improve-codebase-architecture local_owner: internal-review-high-level @@ -213,21 +255,25 @@ watchlist: local_owner: internal-review-high-level reason: Codebase orientation, module maps, caller maps, and domain vocabulary were extracted into high-level review. - source_family: mattpocock/skills - upstream_id: grill-with-docs - local_owner: internal-gateway-writing-plans - reason: Glossary and ADR side effects are not default repository conventions. + upstream_id: prototype + local_owner: unmanaged + reason: Referenced by wayfinder but outside the user-approved import scope; keep visible until explicitly approved or mapped to a local owner. + - source_family: mattpocock/skills + upstream_id: triage + local_owner: unmanaged + reason: Referenced by setup-matt-pocock-skills but not installed in this repository; keep visible until explicitly approved or mapped to a local owner. + - source_family: mattpocock/skills + upstream_id: to-tickets + local_owner: unmanaged + reason: Referenced by setup-matt-pocock-skills but not installed in this repository; keep visible until explicitly approved or mapped to a local owner. - source_family: mattpocock/skills - upstream_id: setup-matt-pocock-skills - local_owner: local-agent-sync-external-resources - reason: Setup conventions are not part of the managed source catalog. + upstream_id: qa + local_owner: unmanaged + reason: Referenced by setup-matt-pocock-skills but not installed in this repository; keep visible until explicitly approved or mapped to a local owner. - source_family: mattpocock/skills upstream_id: caveman local_owner: retired reason: The previously retained caveman import was retired from the managed catalog by explicit user scope reduction. Do not reimport unless the user re-approves the retained import. - - source_family: mattpocock/skills - upstream_id: code-review - local_owner: retired - reason: The mattpocock-code-review import was replaced by addyosmani-code-review-and-quality from addyosmani/agent-skills. Do not reimport unless the user re-approves the retained import. - source_family: addyosmani/agent-skills upstream_id: idea-refine local_owner: internal-gateway-idea diff --git a/.github/skills/local-agent-sync-external-resources/scripts/source_prepare_core.py b/.github/skills/local-agent-sync-external-resources/scripts/source_prepare_core.py new file mode 100644 index 00000000..65956e40 --- /dev/null +++ b/.github/skills/local-agent-sync-external-resources/scripts/source_prepare_core.py @@ -0,0 +1,457 @@ +from __future__ import annotations + +import fcntl +import hashlib +import io +import os +import shutil +import subprocess +import tarfile +import tempfile +import time +from dataclasses import dataclass +from pathlib import Path +from typing import Literal + +from sync_external_resources_core import ( + ManagedResources, + ManagedSource, + SyncCommandError, + _run_command, +) + + +NETWORK_COMMAND_TIMEOUT_SECONDS = 1800 + + +@dataclass(frozen=True) +class PrepareSourceResult: + source_id: str + repository: str + ref: str + cache_status: Literal["cached", "fetched", "rebuilt"] + fetch_strategy: Literal["cache", "direct-sha", "advertised-ref"] + materialized_files: int + materialized_bytes: int + cache_bytes_added: int + duration_ms: int + + +def _cache_key_for_repository(repository: str) -> str: + return hashlib.sha256(repository.encode("utf-8")).hexdigest() + + +def _build_fetch_command(sha: str) -> list[str]: + return [ + "git", + "-c", + "fetch.fsckObjects=true", + "fetch", + "origin", + sha, + "--no-tags", + "--no-recurse-submodules", + "--no-write-fetch-head", + "--filter=blob:none", + "--refmap=", + ] + + +_FORBIDDEN_UPSTREAM_CHARS = {"\\", "\x00"} + + +def _validate_upstream_paths(paths: list[str] | tuple[str, ...]) -> None: + if not paths: + raise ValueError("upstream paths must not be empty") + + normalized: list[str] = [] + for path in paths: + if not path: + raise ValueError("upstream path must not be empty") + if path in (".", ".."): + raise ValueError(f"upstream path must not be . or ..: {path!r}") + if path.startswith("/"): + raise ValueError(f"upstream path must not be absolute: {path!r}") + if any(ch in path for ch in _FORBIDDEN_UPSTREAM_CHARS): + raise ValueError( + f"upstream path contains forbidden character: {path!r}" + ) + parts = path.split("/") + if ".." in parts: + raise ValueError( + f"upstream path must not contain ..: {path!r}" + ) + normalized.append(path) + + seen: set[str] = set() + for path in sorted(normalized): + for existing in seen: + if path.startswith(existing + "/"): + raise ValueError( + f"overlapping upstream paths: {existing!r} and {path!r}" + ) + if path in seen: + raise ValueError(f"duplicate upstream path: {path!r}") + seen.add(path) + + +def _cache_dir(workspace: Path, repository: str) -> Path: + key = _cache_key_for_repository(repository) + return workspace / "cache" / "repositories" / key + + +def _lock_path(cache: Path) -> Path: + return cache.parent / f"{cache.name}.lock" + + +def _acquire_lock(lock_file: Path) -> int: + lock_file.parent.mkdir(parents=True, exist_ok=True) + fd = os.open(str(lock_file), os.O_CREAT | os.O_RDWR) + fcntl.flock(fd, fcntl.LOCK_EX) + return fd + + +def _release_lock(fd: int, lock_file: Path) -> None: + try: + fcntl.flock(fd, fcntl.LOCK_UN) + finally: + os.close(fd) + + +def _init_bare_cache(cache: Path, repository: str) -> None: + cache.mkdir(parents=True, exist_ok=True) + _run_command(["git", "init", "--bare"], cwd=cache) + _run_command( + ["git", "config", "remote.origin.url", repository], + cwd=cache, + ) + _run_command( + ["git", "config", "remote.origin.fetch", ""], + cwd=cache, + ) + _run_command( + ["git", "config", "remote.origin.tagOpt", "--no-tags"], + cwd=cache, + ) + _run_command( + ["git", "config", "core.repositoryFormatVersion", "1"], + cwd=cache, + ) + _run_command( + ["git", "config", "extensions.partialClone", "origin"], + cwd=cache, + ) + _run_command( + ["git", "config", "remote.origin.promisor", "true"], + cwd=cache, + ) + + +def _pin_ref(sha: str) -> str: + return f"refs/cache/pins/{sha}" + + +def _has_pin(cache: Path, sha: str) -> bool: + pin = _pin_ref(sha) + result = subprocess.run( + ["git", "rev-parse", "--verify", pin], + cwd=cache, + capture_output=True, + text=True, + check=False, + ) + if result.returncode != 0: + return False + return result.stdout.strip() == sha + + +def _verify_commit(cache: Path, sha: str) -> None: + result = _run_command( + ["git", "rev-parse", "--verify", f"{sha}^{{commit}}"], + cwd=cache, + ) + resolved = result.stdout.strip() + if resolved != sha: + raise ValueError( + f"commit verification mismatch: expected {sha}, got {resolved}" + ) + type_result = _run_command( + ["git", "cat-file", "-t", sha], + cwd=cache, + ) + obj_type = type_result.stdout.strip() + if obj_type != "commit": + raise ValueError(f"object {sha} is {obj_type}, expected commit") + + +def _write_pin(cache: Path, sha: str) -> None: + _run_command( + ["git", "update-ref", _pin_ref(sha), sha], + cwd=cache, + ) + + +def _fetch_sha(cache: Path, sha: str) -> None: + cmd = _build_fetch_command(sha) + _run_command(cmd, cwd=cache, timeout=NETWORK_COMMAND_TIMEOUT_SECONDS) + + +def _fetch_advertised_ref(cache: Path, ref: str) -> None: + cmd = [ + "git", + "-c", + "fetch.fsckObjects=true", + "fetch", + "origin", + ref, + "--no-tags", + "--no-recurse-submodules", + "--no-write-fetch-head", + "--filter=blob:none", + "--refmap=", + ] + _run_command(cmd, cwd=cache, timeout=NETWORK_COMMAND_TIMEOUT_SECONDS) + + +def _cache_size(cache: Path) -> int: + total = 0 + objects_dir = cache / "objects" + if not objects_dir.exists(): + return 0 + for entry in objects_dir.rglob("*"): + if entry.is_file() and not entry.name.endswith(".lock"): + total += entry.stat().st_size + return total + + +def _fetch_source( + cache: Path, + source: ManagedSource, +) -> tuple[Literal["cached", "fetched"], Literal["cache", "direct-sha", "advertised-ref"]]: + if _has_pin(cache, source.ref): + return "cached", "cache" + + fetch_strategy: Literal["direct-sha", "advertised-ref"] = "direct-sha" + try: + _fetch_sha(cache, source.ref) + except SyncCommandError: + if source.advertised_ref is None: + raise + _fetch_advertised_ref(cache, source.advertised_ref) + fetch_strategy = "advertised-ref" + + _verify_commit(cache, source.ref) + _write_pin(cache, source.ref) + + return "fetched", fetch_strategy + + +def _safe_filter(member: tarfile.TarInfo, dest_path: str) -> tarfile.TarInfo: + if member.name.startswith("/"): + raise tarfile.FilterError(f"absolute path in tar member: {member.name!r}") + result = tarfile.data_filter(member, dest_path) + return result + + +def _extract_archive(archive_bytes: bytes, export_dir: Path) -> None: + with tarfile.open(fileobj=io.BytesIO(archive_bytes)) as tar: + tar.extractall(path=export_dir, filter=_safe_filter) + + +def _export_paths( + cache: Path, + source: ManagedSource, + export_dir: Path, +) -> tuple[int, int]: + upstream_paths = [asset.upstream for asset in source.assets] + + cmd = [ + "git", + "-C", + str(cache), + "archive", + source.ref, + "--", + *upstream_paths, + ] + result = subprocess.run( + cmd, + capture_output=True, + check=False, + ) + if result.returncode != 0: + raise SyncCommandError( + cmd, + result.returncode, + result.stderr.decode("utf-8", errors="replace")[:500], + ) + + _extract_archive(result.stdout, export_dir) + + files_count = 0 + bytes_count = 0 + for entry in export_dir.rglob("*"): + if entry.is_file(): + files_count += 1 + bytes_count += entry.lstat().st_size + elif entry.is_symlink(): + files_count += 1 + bytes_count += len(os.readlink(entry)) + + return files_count, bytes_count + + +def _write_source_metadata( + snapshot: Path, + source: ManagedSource, +) -> None: + upstream_paths = sorted(asset.upstream for asset in source.assets) + digest = hashlib.sha256( + ",".join(upstream_paths).encode("utf-8") + ).hexdigest() + tsv_content = ( + f"source_id\trepository\tref\tpaths_sha256\n" + f"{source.source_id}\t{source.repository}\t{source.ref}\t{digest}\n" + ) + (snapshot / ".external-resource-source.tsv").write_text( + tsv_content, encoding="utf-8" + ) + + +def _publish_snapshot_atomic( + sources_root: Path, + source_id: str, + staging: Path, + source: ManagedSource, +) -> None: + target = sources_root / source_id + prior = target.parent / f"{target.name}.prior" + + _write_source_metadata(staging, source) + + if target.exists(): + if prior.exists(): + shutil.rmtree(prior) + target.rename(prior) + + try: + staging.rename(target) + except Exception: + if prior.exists(): + prior.rename(target) + raise + + if prior.exists(): + shutil.rmtree(prior) + + +def _rebuild_cache_beside( + cache: Path, + source: ManagedSource, +) -> tuple[Literal["direct-sha", "advertised-ref"], int]: + staging = cache.parent / f"{cache.name}.rebuild" + prior = cache.parent / f"{cache.name}.prior" + if staging.exists(): + shutil.rmtree(staging) + if prior.exists(): + shutil.rmtree(prior) + + _init_bare_cache(staging, source.repository) + _, fetch_strategy = _fetch_source(staging, source) + if fetch_strategy == "cache": + raise ValueError("rebuilt cache must perform a fetch") + rebuilt_bytes = _cache_size(staging) + + if cache.exists(): + cache.rename(prior) + try: + staging.rename(cache) + except Exception: + if prior.exists(): + prior.rename(cache) + raise + if prior.exists(): + shutil.rmtree(prior) + + return fetch_strategy, rebuilt_bytes + + +def _prepare_one_source( + source: ManagedSource, + workspace: Path, + sources_root: Path, + rebuild_cache: bool, +) -> PrepareSourceResult: + start = time.monotonic() + cache = _cache_dir(workspace, source.repository) + lock_file = _lock_path(cache) + fd = _acquire_lock(lock_file) + + try: + needs_init = not (cache / "HEAD").exists() + if needs_init: + _init_bare_cache(cache, source.repository) + + if rebuild_cache and (cache / "HEAD").exists(): + fetch_strategy, rebuilt_bytes = _rebuild_cache_beside(cache, source) + cache_status: Literal["cached", "fetched", "rebuilt"] = "rebuilt" + bytes_added = rebuilt_bytes + else: + before = _cache_size(cache) + cache_status, fetch_strategy = _fetch_source(cache, source) + bytes_added = max(0, _cache_size(cache) - before) + + staging_dir = sources_root.parent / f".{source.source_id}.staging" + if staging_dir.exists(): + shutil.rmtree(staging_dir) + staging_dir.mkdir(parents=True) + + try: + files_count, bytes_count = _export_paths( + cache, source, staging_dir + ) + _publish_snapshot_atomic( + sources_root, source.source_id, staging_dir, source + ) + except Exception: + if staging_dir.exists(): + shutil.rmtree(staging_dir) + raise + + finally: + _release_lock(fd, lock_file) + + elapsed_ms = int((time.monotonic() - start) * 1000) + + return PrepareSourceResult( + source_id=source.source_id, + repository=source.repository, + ref=source.ref, + cache_status=cache_status, + fetch_strategy=fetch_strategy, + materialized_files=files_count, + materialized_bytes=bytes_count, + cache_bytes_added=bytes_added, + duration_ms=elapsed_ms, + ) + + +def prepare_sources( + resources: ManagedResources, + workspace: Path, + sources_root: Path, + *, + rebuild_cache: bool = False, +) -> tuple[PrepareSourceResult, ...]: + sources_root.mkdir(parents=True, exist_ok=True) + results: list[PrepareSourceResult] = [] + for source in resources.sources: + _validate_upstream_paths( + [asset.upstream for asset in source.assets] + ) + result = _prepare_one_source( + source, workspace, sources_root, rebuild_cache + ) + results.append(result) + return tuple(results) + diff --git a/.github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py b/.github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py index 93bc1684..8a58b94b 100644 --- a/.github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py +++ b/.github/skills/local-agent-sync-external-resources/scripts/sync_external_resources.py @@ -8,7 +8,7 @@ import subprocess import sys import tempfile -from dataclasses import dataclass, field +from dataclasses import dataclass from pathlib import Path from typing import Literal, Sequence @@ -29,10 +29,18 @@ validate_external_workspace, validate_override_patches, ) +from sync_output_core import ( # noqa: E402 + OutputRecord, + render_tsv, +) DEFAULT_MANIFEST = ( SCRIPT_DIR.parent / "references" / "managed-resources.yaml" ).as_posix() +from source_prepare_core import ( # noqa: E402 + prepare_sources, +) + DEFAULT_OVERRIDES = ( SCRIPT_DIR.parent / "references" / "imported-asset-overrides.yaml" ).as_posix() @@ -40,7 +48,7 @@ @dataclass(frozen=True) class SyncOutcome: - mode: Literal["audit", "plan", "apply"] + mode: Literal["prepare", "audit", "plan", "apply"] workspace: str | None managed_assets: int changed_paths: tuple[str, ...] @@ -48,9 +56,10 @@ class SyncOutcome: validations: tuple[str, ...] blockers: tuple[str, ...] repository_changed: bool + source_results: tuple[object, ...] = () def to_dict(self) -> dict[str, object]: - return { + result: dict[str, object] = { "mode": self.mode, "workspace": self.workspace, "managed_assets": self.managed_assets, @@ -67,13 +76,118 @@ def to_dict(self) -> dict[str, object]: "blockers": list(self.blockers), "repository_changed": self.repository_changed, } + if self.source_results: + result["source_results"] = [ + { + "source_id": r.source_id, + "repository": r.repository, + "ref": r.ref, + "cache_status": r.cache_status, + "fetch_strategy": r.fetch_strategy, + "materialized_files": r.materialized_files, + "materialized_bytes": r.materialized_bytes, + "cache_bytes_added": r.cache_bytes_added, + "duration_ms": r.duration_ms, + } + for r in self.source_results + ] + return result + + def to_records(self) -> tuple[OutputRecord, ...]: + records: list[OutputRecord] = [] + records.append( + OutputRecord("summary", "mode", "ok", self.mode) + ) + records.append( + OutputRecord( + "summary", "managed_assets", "ok", str(self.managed_assets) + ) + ) + records.append( + OutputRecord( + "summary", "changed_paths", "ok", str(len(self.changed_paths)) + ) + ) + records.append( + OutputRecord( + "summary", + "override_results", + "ok", + str(len(self.override_results)), + ) + ) + records.append( + OutputRecord( + "summary", + "repository_changed", + "ok", + str(self.repository_changed).lower(), + ) + ) + if self.workspace is not None: + records.append( + OutputRecord("summary", "workspace", "ok", self.workspace) + ) + for validation in self.validations: + records.append( + OutputRecord("validation", validation, "ok", "") + ) + for blocker in self.blockers: + records.append( + OutputRecord("blocker", blocker, "fail", "") + ) + for path in self.changed_paths: + records.append( + OutputRecord("change", path, "ok", "") + ) + for result in self.override_results: + records.append( + OutputRecord( + "override", + result.override_id, + result.status, + result.target_path, + ) + ) + for sr in self.source_results: + records.append( + OutputRecord( + "source", + sr.source_id, + sr.cache_status, + sr.ref, + ) + ) + for metric_name, metric_value in ( + ("materialized_files", sr.materialized_files), + ("materialized_bytes", sr.materialized_bytes), + ("cache_bytes_added", sr.cache_bytes_added), + ("duration_ms", sr.duration_ms), + ): + records.append( + OutputRecord( + "metric", + f"{sr.source_id}.{metric_name}", + "ok", + str(metric_value), + ) + ) + records.append( + OutputRecord( + "validation", + f"{sr.source_id}.fetch_strategy", + "ok", + sr.fetch_strategy, + ) + ) + return tuple(records) def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser( description="Audit, plan, or apply declared external resource refreshes." ) - parser.add_argument("mode", choices=("audit", "plan", "apply")) + parser.add_argument("mode", choices=("prepare", "audit", "plan", "apply")) parser.add_argument("--repo-root", default=".") parser.add_argument("--workspace") parser.add_argument("--manifest", default=DEFAULT_MANIFEST) @@ -83,7 +197,12 @@ def build_parser() -> argparse.ArgumentParser: help="Use prepared source checkouts instead of network fetch.", ) parser.add_argument("--allow-dirty", action="store_true") - parser.add_argument("--format", choices=("text", "json"), default="text") + parser.add_argument("--format", choices=("text", "tsv", "json"), default="text") + parser.add_argument( + "--rebuild-cache", + action="store_true", + help="Force rebuild of the Git object cache (prepare mode only).", + ) return parser @@ -261,6 +380,35 @@ def _apply_candidate_patch( ) +def _prepare( + repo_root: Path, + workspace: Path, + resources: ManagedResources, + rebuild_cache: bool, +) -> SyncOutcome: + validate_external_workspace(repo_root, workspace) + sources_root = workspace / "sources" + + results = prepare_sources( + resources, + workspace, + sources_root, + rebuild_cache=rebuild_cache, + ) + + return SyncOutcome( + mode="prepare", + workspace=str(workspace), + managed_assets=len(resources.assets), + changed_paths=(), + override_results=(), + validations=("manifest-pins-validated", "sources-prepared"), + blockers=(), + repository_changed=False, + source_results=results, + ) + + def _audit( repo_root: Path, resources: ManagedResources, @@ -398,6 +546,25 @@ def _format_text(outcome: SyncOutcome) -> str: return "\n".join(lines) +def _requested_format(argv: Sequence[str] | None) -> str: + values = list(argv) if argv is not None else sys.argv[1:] + for index, value in enumerate(values): + if value == "--format" and index + 1 < len(values): + return values[index + 1] + if value.startswith("--format="): + return value.split("=", 1)[1] + return "text" + + +def _emit_failure(fmt: str, message: str) -> None: + if fmt == "json": + print(json.dumps({"blockers": [message], "repository_changed": False}, indent=2)) + elif fmt == "tsv": + print(render_tsv((OutputRecord("blocker", message, "fail", ""),)), end="") + else: + print(f"Blockers: {message}") + + def run(argv: Sequence[str] | None = None) -> int: parser = build_parser() args = parser.parse_args(argv) @@ -412,11 +579,24 @@ def run(argv: Sequence[str] | None = None) -> int: resources = load_managed_resources(manifest_path) - if args.mode == "audit": + if args.mode == "prepare": + if not args.workspace: + parser.error("prepare mode requires --workspace") + outcome = _prepare( + repo_root, + Path(args.workspace).resolve(), + resources, + args.rebuild_cache, + ) + elif args.mode == "audit": + if args.rebuild_cache: + parser.error("--rebuild-cache is only valid for prepare mode") outcome = _audit(repo_root, resources, overrides_path) elif args.mode == "plan": if not args.workspace: parser.error("plan mode requires --workspace") + if args.rebuild_cache: + parser.error("--rebuild-cache is only valid for prepare mode") outcome = _plan( repo_root, Path(args.workspace).resolve(), @@ -427,6 +607,8 @@ def run(argv: Sequence[str] | None = None) -> int: elif args.mode == "apply": if not args.workspace: parser.error("apply mode requires --workspace") + if args.rebuild_cache: + parser.error("--rebuild-cache is only valid for prepare mode") outcome = _apply( repo_root, Path(args.workspace).resolve(), @@ -440,18 +622,26 @@ def run(argv: Sequence[str] | None = None) -> int: if args.format == "json": print(json.dumps(outcome.to_dict(), indent=2)) + elif args.format == "tsv": + print(render_tsv(outcome.to_records()), end="") else: print(_format_text(outcome)) if args.mode == "audit": return 0 + if args.mode == "prepare": + return 0 if outcome.blockers: return 2 return 0 def main(argv: Sequence[str] | None = None) -> int: - return run(argv) + try: + return run(argv) + except (ValueError, SyncCommandError) as exc: + _emit_failure(_requested_format(argv), str(exc)) + return 2 if __name__ == "__main__": diff --git a/.github/skills/local-agent-sync-external-resources/scripts/sync_external_resources_core.py b/.github/skills/local-agent-sync-external-resources/scripts/sync_external_resources_core.py index f4ac5efa..baf327c0 100644 --- a/.github/skills/local-agent-sync-external-resources/scripts/sync_external_resources_core.py +++ b/.github/skills/local-agent-sync-external-resources/scripts/sync_external_resources_core.py @@ -12,6 +12,9 @@ import yaml +_COMMIT_OBJECT_ID_RE = re.compile(r"^(?:[0-9a-f]{40}|[0-9a-f]{64})$") + + @dataclass(frozen=True) class ManagedAsset: source: str @@ -25,7 +28,11 @@ class ManagedSource: source_id: str repository: str ref: str + advertised_ref: str | None assets: tuple[ManagedAsset, ...] + rewrite_skill_references: bool = False + skill_reference_aliases: tuple[tuple[str, str], ...] = () + backtick_skill_references: tuple[str, ...] = () @dataclass(frozen=True) @@ -60,6 +67,23 @@ def _require_non_empty_string(value: object, field: str) -> str: return value.strip() +def _optional_non_empty_string(value: object, field: str) -> str | None: + if value is None: + return None + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{field} must be a non-empty string when provided.") + return value.strip() + + +def _require_commit_object_id(value: str, field: str) -> str: + stripped = value.strip() + if not _COMMIT_OBJECT_ID_RE.match(stripped): + raise ValueError( + f"{field} must be a full lowercase commit object ID, got {stripped!r}." + ) + return stripped + + def load_managed_resources(path: Path) -> ManagedResources: payload = yaml.safe_load(path.read_text(encoding="utf-8")) if not isinstance(payload, dict) or payload.get("version") != 1: @@ -89,8 +113,15 @@ def load_managed_resources(path: Path) -> ManagedResources: repository = _require_non_empty_string( raw_source.get("repository"), f"source {source_id} repository" ) - ref = _require_non_empty_string( - raw_source.get("ref"), f"source {source_id} ref" + ref = _require_commit_object_id( + _require_non_empty_string( + raw_source.get("ref"), f"source {source_id} ref" + ), + f"source {source_id} ref", + ) + advertised_ref = _optional_non_empty_string( + raw_source.get("advertised_ref"), + f"source {source_id} advertised_ref", ) raw_assets = raw_source.get("assets") if not isinstance(raw_assets, list) or not raw_assets: @@ -134,12 +165,62 @@ def load_managed_resources(path: Path) -> ManagedResources: ) ) + rewrite_skill_references = raw_source.get("rewrite_skill_references", False) + if not isinstance(rewrite_skill_references, bool): + raise ValueError( + f"source {source_id} rewrite_skill_references must be a boolean." + ) + raw_skill_reference_aliases = raw_source.get("skill_reference_aliases", {}) + if not isinstance(raw_skill_reference_aliases, dict): + raise ValueError( + f"source {source_id} skill_reference_aliases must be a mapping." + ) + canonical_names = {asset.canonical_name for asset in assets} + skill_reference_aliases: list[tuple[str, str]] = [] + for alias, canonical_name in raw_skill_reference_aliases.items(): + alias = _require_non_empty_string( + alias, f"source {source_id} skill reference alias" + ) + canonical_name = _require_non_empty_string( + canonical_name, + f"source {source_id} skill reference alias target", + ) + if canonical_name not in canonical_names: + raise ValueError( + f"source {source_id} skill reference alias target " + f"{canonical_name} is not a declared asset canonical name." + ) + skill_reference_aliases.append((alias, canonical_name)) + + raw_backtick_refs = raw_source.get("backtick_skill_references", []) + if not isinstance(raw_backtick_refs, list): + raise ValueError( + f"source {source_id} backtick_skill_references must be a list." + ) + upstream_basenames = {Path(asset.upstream).name for asset in assets} + alias_names = {alias for alias, _ in skill_reference_aliases} + backtick_skill_references: list[str] = [] + for raw_backtick_ref in raw_backtick_refs: + backtick_ref = _require_non_empty_string( + raw_backtick_ref, f"source {source_id} backtick skill reference" + ) + if backtick_ref not in upstream_basenames and backtick_ref not in alias_names: + raise ValueError( + f"source {source_id} backtick skill reference {backtick_ref} " + "is not a declared upstream asset basename or alias." + ) + backtick_skill_references.append(backtick_ref) + sources.append( ManagedSource( source_id=source_id, repository=repository, ref=ref, + advertised_ref=advertised_ref, assets=tuple(assets), + rewrite_skill_references=rewrite_skill_references, + skill_reference_aliases=tuple(skill_reference_aliases), + backtick_skill_references=tuple(backtick_skill_references), ) ) @@ -207,14 +288,21 @@ def __init__(self, command: list[str], exit_code: int, stderr: str) -> None: ) -def _run_command(command: list[str], cwd: Path | None = None) -> subprocess.CompletedProcess[str]: +LOCAL_COMMAND_TIMEOUT_SECONDS = 60 + + +def _run_command( + command: list[str], + cwd: Path | None = None, + timeout: int = LOCAL_COMMAND_TIMEOUT_SECONDS, +) -> subprocess.CompletedProcess[str]: result = subprocess.run( command, cwd=cwd, check=False, capture_output=True, text=True, - timeout=60, + timeout=timeout, ) if result.returncode != 0: raise SyncCommandError(command, result.returncode, result.stderr) @@ -242,13 +330,19 @@ def find_dirty_targets( return () result = _run_git( repo_root, - ["status", "--porcelain=v1", "--", *(asset.local for asset in assets)], - ) - return tuple( - line[3:] - for line in result.stdout.splitlines() - if len(line) > 3 + ["status", "--porcelain=v1", "-z", "--", *(asset.local for asset in assets)], ) + fields = [field for field in result.stdout.split("\0") if field] + dirty: list[str] = [] + index = 0 + while index < len(fields): + entry = fields[index] + status_code = entry[:2] + dirty.append(entry[3:]) + if status_code[0] in {"R", "C"}: + index += 1 + index += 1 + return tuple(dirty) def collect_missing_upstream_paths( @@ -265,6 +359,27 @@ def collect_missing_upstream_paths( return tuple(missing) +def validate_prepared_sources( + resources: ManagedResources, + sources_root: Path, +) -> None: + missing: list[str] = [] + for source in resources.sources: + metadata = ( + sources_root / source.source_id / ".external-resource-source.tsv" + ) + if not metadata.exists(): + missing.append(source.source_id) + if missing: + raise ValueError( + "Missing prepared source metadata: " + + ", ".join(missing) + + ". Expected prepared sources under " + + sources_root.as_posix() + + ". Run prepare before audit/plan/apply." + ) + + def materialize_candidate( resources: ManagedResources, workspace: Path, @@ -278,6 +393,8 @@ def materialize_candidate( if sources_root is None: sources_root = workspace / "sources" + validate_prepared_sources(resources, sources_root) + missing = collect_missing_upstream_paths(resources, sources_root) if missing: details = "; ".join(missing) @@ -301,6 +418,46 @@ def materialize_candidate( _FRONTMATTER_NAME_RE = re.compile(r"^(name\s*:\s*).*$", re.MULTILINE) _SUPERPOWERS_SKILL_REF_RE = re.compile(r"\bsuperpowers:([a-z0-9][a-z0-9-]*)") +_SLASH_SKILL_REF_RE = re.compile( + r"(?[a-z0-9][a-z0-9-]*)\b" +) +_BACKTICK_SKILL_REF_RE = re.compile( + r"(?[a-z0-9][a-z0-9-]*)`" +) +_GUIDED_QUESTION_SKILLS = frozenset({"superpowers-brainstorming", "grill-me"}) +_GUIDED_QUESTION_CONTRACT_START = "" +_GUIDED_QUESTION_CONTRACT_END = "" +_GUIDED_QUESTION_CONTRACT_RE = re.compile( + re.escape(_GUIDED_QUESTION_CONTRACT_START) + + r".*?" + + re.escape(_GUIDED_QUESTION_CONTRACT_END), + re.DOTALL, +) +_GUIDED_QUESTION_CONTRACT = f"""\ +{_GUIDED_QUESTION_CONTRACT_START} +## Local guided-question contract + +This repository-owned contract overrides any earlier instruction to ask one question at a time. + +- Ask all currently known questions in numbered bulk question blocks. +- Use `Question`, `Recommendation`, `Why`, and `Default if accepted` for every + numbered question. +- Make `Recommendation` the suggested answer and `Why` its concrete rationale. +- Keep each question, recommendation, and reason brief, clear, and + decision-ready. +- Put unresolved follow-ups in another numbered block. If only one blocking + question remains, present it as a numbered one-item block. +{_GUIDED_QUESTION_CONTRACT_END}""" + + +def _enforce_guided_question_contract(content: str) -> str: + if _GUIDED_QUESTION_CONTRACT_RE.search(content): + return _GUIDED_QUESTION_CONTRACT_RE.sub( + _GUIDED_QUESTION_CONTRACT, + content, + count=1, + ) + return content.rstrip() + "\n\n" + _GUIDED_QUESTION_CONTRACT + "\n" def normalize_candidate( @@ -312,6 +469,24 @@ def normalize_candidate( replacements_by_source.setdefault(replacement.source, []).append(replacement) changed: list[str] = [] + skill_references_by_source: dict[str, dict[str, str]] = {} + backtick_references_by_source: dict[str, dict[str, str]] = {} + for source in resources.sources: + if not source.rewrite_skill_references: + continue + skill_references = { + Path(asset.upstream).name: asset.canonical_name + for asset in source.assets + } + skill_references.update(dict(source.skill_reference_aliases)) + skill_references_by_source[source.source_id] = skill_references + declared_backticks = set(source.backtick_skill_references) + backtick_references_by_source[source.source_id] = { + name: canonical + for name, canonical in skill_references.items() + if name in declared_backticks + } + for asset in resources.assets: asset_dir = candidate / asset.local if not asset_dir.exists(): @@ -334,6 +509,32 @@ def normalize_candidate( r"superpowers-\1", content ) + skill_references = skill_references_by_source.get(asset.source, {}) + if skill_references: + content = _SLASH_SKILL_REF_RE.sub( + lambda match: "/" + + skill_references.get( + match.group("name"), match.group("name") + ), + content, + ) + backtick_references = backtick_references_by_source.get(asset.source, {}) + if backtick_references: + content = _BACKTICK_SKILL_REF_RE.sub( + lambda match: "`" + + backtick_references.get( + match.group("name"), match.group("name") + ) + + "`", + content, + ) + + if ( + asset.canonical_name in _GUIDED_QUESTION_SKILLS + and file_path == asset_dir / "SKILL.md" + ): + content = _enforce_guided_question_contract(content) + for replacement in replacements_by_source.get(asset.source, []): content = content.replace(replacement.old, replacement.new) diff --git a/.github/skills/local-agent-sync-external-resources/scripts/sync_output_core.py b/.github/skills/local-agent-sync-external-resources/scripts/sync_output_core.py new file mode 100644 index 00000000..dd45ef27 --- /dev/null +++ b/.github/skills/local-agent-sync-external-resources/scripts/sync_output_core.py @@ -0,0 +1,39 @@ +from __future__ import annotations + +from dataclasses import dataclass + + +@dataclass(frozen=True, order=True) +class OutputRecord: + record: str + key: str + status: str + value: str + + +def escape_tsv(value: str) -> str: + return ( + value + .replace("\\", "\\\\") + .replace("\t", "\\t") + .replace("\n", "\\n") + .replace("\r", "\\r") + ) + + +def render_tsv(records: tuple[OutputRecord, ...] | list[OutputRecord]) -> str: + sorted_records = sorted(records) + lines = ["record\tkey\tstatus\tvalue"] + for record in sorted_records: + lines.append( + "\t".join( + escape_tsv(field) + for field in ( + record.record, + record.key, + record.status, + record.value, + ) + ) + ) + return "\n".join(lines) + "\n" diff --git a/.github/skills/local-agent-sync-global-copilot-configs-into-repo/SKILL.md b/.github/skills/local-agent-sync-global-copilot-configs-into-repo/SKILL.md deleted file mode 100644 index 5b93d190..00000000 --- a/.github/skills/local-agent-sync-global-copilot-configs-into-repo/SKILL.md +++ /dev/null @@ -1,94 +0,0 @@ ---- -name: local-agent-sync-global-copilot-configs-into-repo -description: Use when aligning a consumer repository to this repository's managed GitHub Copilot baseline, shared repository-hygiene files, and retained-learning ledger template. ---- - -# Internal Agent Sync Global Copilot Configs Into Repo - -## Referenced skills - -- None. - -Use this skill as the mandatory operating engine for `.github/agents/local-sync-global-copilot-configs-into-repo.agent.md`. - -This skill owns the reusable sync procedure. Keep the paired agent short; do not duplicate the analyze, plan, apply, reporting, or automation rules there. - -The paired agent should not restate default mode handling, preserved `local-*` behavior, `internal-sync-*` exclusions, plan-file lifecycle, or automation entrypoints from this skill. - -## When to use - -- Align a consumer repository with the managed GitHub Copilot baseline from this repository. -- Refresh target `AGENTS.md`, `.github/copilot-instructions.md`, and `.github/INVENTORY.md` to the current root-policy and review-only model after mirroring. -- Refresh shared repository-hygiene files that are part of the managed sync baseline, currently `.editorconfig`, `.pre-commit-config.yaml`, `.github/workflows/_pre-commit.yml`, `.github/copilot-commit-message-instructions.md`, `.github/security-baseline.md`, `.github/DEPRECATION.md`, and `.github/repo-profiles.yml`. -- Refresh repository-root `LESSONS_LEARNED.md` from the source structure while preserving and, when needed, migrating target-authored pending lesson rows. -- Run or interpret `./.github/scripts/run.sh sync_copilot_catalog` or `.github/scripts/sync_copilot_catalog.py`. -- Audit source-target drift before or after a sync. - -## Core Operating Contract - -- Treat this repository as the source of truth. -- Keep target assumptions narrow: GitHub Copilot assets live under `.github/` and `AGENTS.md` stays at repository root. -- Preserve target `local-*` assets under mirrored categories and delete target-only non-local assets there during `apply`. -- When consumer-local creator bundles depend on shared runtime-critical rules, mirror those rules inside each creator bundle as source-managed files and keep the mirror paths registered in the source inventory and target manifest; do not rely on cross-bundle references or unsynced local-only resources for creator runtime behavior. -- When the source baseline includes an approved imported-asset override registry plus replay patches, mirror that governance bundle as source-managed state instead of recreating target-local hidden forks on imported assets. -- Exclude source resources named `internal-sync-*` from consumer mirroring and remove any target copies of those resources during `apply`. -- Create consumer-local `docs/README.md`, `docs/repository-context.md`, `docs/architecture.md`, `docs/tech.md`, and `docs/structure.md` from `.github/templates/` only when missing, then preserve target-authored content on later sync runs. -- Delete retired standalone runtime operating model documents from consumers; runtime workflow guidance now travels through root guidance and skills. -- Keep root guidance layered: `AGENTS.md` is the agent policy entrypoint, `.github/copilot-instructions.md` is review-only for GitHub.com Copilot code review, and `.github/INVENTORY.md` is the live catalog. -- Treat `LESSONS_LEARNED.md` as a source-managed retained-learning template: create it when missing, keep its structure aligned with the source contract, and preserve target-authored pending lessons instead of overwriting them with source rows. -- Mirror only the explicitly shared repository-hygiene files declared in `references/sync-contract.md`; do not widen workflow or root-file mirroring implicitly. -- Ensure the target repository `.gitignore` contains an ignore rule for `tmp/superpowers/`. -- Treat `.vscode/settings.json` as consumer-owned JSONC and manage only the Copilot settings required to disable instruction-file loading. -- When moving from `plan` to `apply` against the same target, pass `--allow-dirty-target` only when the generated `tmp/copilot-sync.plan.md` is the sole target diff left by the planning run. -- Prefer the bundled sync automation when it matches the requested mode instead of re-deriving the workflow manually. -- Keep detailed operating rules in `references/sync-contract.md` instead of re-expanding them in the agent body. - -## Default Workflow - -1. Analyze the source baseline, target catalog, target git state, and preserved local assets. -2. Write `tmp/copilot-sync.plan.md` in the target repository with the pending operations and any manual follow-up that remains outside automation. -3. In `apply`, mirror source-managed assets, merge target `LESSONS_LEARNED.md` rows into the current source structure, rebuild the target inventory, write the target manifest, and clear the tracking plan when nothing remains pending. -4. Re-run the closest existing validation and report any blockers or gaps. - -## Mode Selection - -- `plan`: default mode and safest starting point. -- `apply`: explicit only, after reviewing a conflict-safe plan and current source findings. -- `audit`: use when source or target drift needs diagnosis before deciding whether to plan or apply; prefer `./.github/scripts/run.sh audit_copilot_catalog` plus the sync planner evidence instead of inventing a third sync mode. - -## Agent-facing output modes - -- For model-facing runs, prefer bounded output over full detail when the script supports it. -- Use `python3 ./.github/scripts/sync_copilot_catalog.py plan --target-repo --format compact` for planner runs, and summarize only status, blockers, warnings, managed mutation counts, and next action in agent responses. -- Use `python3 ./.github/scripts/sync_copilot_catalog.py apply --target-repo --format compact` only after explicit approval, and keep apply reporting bounded to blockers, warnings, changed path evidence, validation status, and next action. -- Reserve full `--format json` output for durable artifacts, audits, debugging, or explicit user request. -- For validator and consistency commands that do not support compact, keep output bounded by using the narrowest target scope and report concise summaries instead of raw log dumps. - -## Evidence Budget - -Collect the minimum evidence set before moving past analysis or approving `apply`: - -- selected mode: `plan`, `apply`, or `audit` -- target git state, including planner-reported relevant `dirty_paths` -- planner output, from `tmp/copilot-sync.plan.md`, JSON output, or both -- preserved target-owned assets covered by the sync contract, including `local-*` assets and consumer-local knowledge documents -- planner-reported `managed_mutation_paths` plus any `dirty_managed_overlap` -- latest validation result for the touched sync behavior - -Keep manual inspection narrow. Review only: - -- paths whose planned action is `create`, `update`, `ensure`, `rebuild`, or `delete` -- dirty paths that overlap planned managed mutations - -If `dirty_managed_overlap` is empty, `--allow-dirty-target` can stay eligible when the other gates are green. If overlap is non-empty, reconcile those paths first or require explicit approval before `apply`. - -## Load On Demand - -- Read `references/sync-contract.md` for exact mirrored categories, exclusions, root-guidance ownership, plan-file lifecycle, automation entrypoints, validation sequence, and reporting requirements. - -## Validation - -- For source-side baseline changes, prefer `./.github/scripts/run.sh check_catalog_consistency --root . --include-token-risks`. -- Rebuild `.github/INVENTORY.md` when touched catalog paths require it by using `./.github/scripts/run.sh build_inventory --root .`. -- For sync automation changes, run `pytest tests/test_sync_and_token_risks.py`. -- If a dedicated sync-contract test does not exist for the touched behavior, say so explicitly and use the closest existing verification. diff --git a/.github/skills/local-agent-sync-global-copilot-configs-into-repo/agents/openai.yaml b/.github/skills/local-agent-sync-global-copilot-configs-into-repo/agents/openai.yaml deleted file mode 100644 index e5b5d0e7..00000000 --- a/.github/skills/local-agent-sync-global-copilot-configs-into-repo/agents/openai.yaml +++ /dev/null @@ -1,4 +0,0 @@ -interface: - display_name: "Copilot Baseline Sync" - short_description: "Mirror the managed Copilot baseline into consumer repos" - default_prompt: "Use $local-agent-sync-global-copilot-configs-into-repo to plan or apply a consumer-repository baseline sync from this standards repository." diff --git a/.github/skills/local-agent-sync-global-copilot-configs-into-repo/references/sync-contract.md b/.github/skills/local-agent-sync-global-copilot-configs-into-repo/references/sync-contract.md deleted file mode 100644 index 17b68f49..00000000 --- a/.github/skills/local-agent-sync-global-copilot-configs-into-repo/references/sync-contract.md +++ /dev/null @@ -1,118 +0,0 @@ -# Sync Contract - -Use this reference when the paired agent or this skill needs the exact sync rules instead of the compact workflow summary. - -## Source-Managed Scope - -Mirror or structurally align these source-managed paths into the consumer repository: - -- `AGENTS.md` -- `LESSONS_LEARNED.md` -- `.editorconfig` -- `.pre-commit-config.yaml` -- `.github/copilot-instructions.md` -- `.github/copilot-commit-message-instructions.md` -- `.github/security-baseline.md` -- `.github/DEPRECATION.md` -- `.github/repo-profiles.yml` -- `.github/workflows/_pre-commit.yml` -- `.github/agents/**` -- `.github/skills/**`, including bundled `references/`, `assets/`, `scripts/`, `agents/`, and licenses - -Apply field-level management to `.vscode/settings.json` only for these keys: - -- `github.copilot.chat.codeGeneration.useInstructionFiles: false` -- `chat.instructionsFilesLocations[".github/instructions"]: false` - -Do not sync `README.md`, changelogs, other workflows, templates, or bootstrap helpers unless the user explicitly expands scope. -Use `.github/templates/` only as standards-repository scaffold source material; do not mirror it into consumer repositories as an operational catalog family. -Do not sync consumer-facing resources whose file or directory name starts with `internal-sync-`; those remain source-only operational controls for the standards repository. -Treat `LESSONS_LEARNED.md` as a structure-managed exception: sync the source template and contract, but preserve target-authored pending lesson rows instead of copying source rows into consumer repositories. -When a consumer-local creator depends on shared runtime-critical rules, keep a self-contained mirror of those rules inside the creator bundle and track the mirror path in the source inventory plus the target `.github/copilot-sync.manifest.json`; do not assume cross-bundle references or target-local extras will be present at runtime. - -## Target Rules - -- Preserve target `local-*` assets under mirrored categories and surface them in the plan or final report. -- Create `docs/README.md`, `docs/repository-context.md`, `docs/architecture.md`, `docs/tech.md`, and `docs/structure.md` from `.github/templates/` only when missing, then preserve them as consumer-local content. -- If a target has legacy `docs/01-local-architecture.md` or `docs/01-architecture.md` and lacks `docs/architecture.md`, rename the legacy file to the canonical path. If canonical and legacy paths coexist, block apply and require manual reconciliation. -- If a target has legacy `docs/02-local-repository-context.md` or `docs/02-repository-context.md` and lacks `docs/repository-context.md`, rename the legacy file to the canonical path. If canonical and legacy paths coexist, block apply and require manual reconciliation. -- Delete legacy `docs/runtime-fit.md` and retired standalone runtime operating model documents. Runtime workflow guidance now lives in root guidance and skills. -- Do not mirror source `.github/instructions/**` files. Delete target non-`local-*` instruction files and preserve target-owned `local-*` instruction files as local exceptions. -- Delete target-owned non-`local-*` assets inside mirrored categories during `apply`. -- Keep the target target-agnostic. The default assumptions are only `.github/` and root `AGENTS.md`. -- Ensure target root `LESSONS_LEARNED.md` exists. If it already exists, align it to the current source structure and migrate preserved pending lesson rows when the source table shape changes. -- Ensure the target root `.gitignore` contains an ignore entry for `tmp/superpowers/`. -- Treat the `.gitignore` update as target-local hygiene: ensure the required ignore entry exists without mirroring the source repository `.gitignore`. -- Keep `.vscode/settings.json` consumer-owned as a file: merge only the two managed Copilot keys, preserve unrelated settings/comments, and block apply with `manual` when malformed JSONC or duplicate relevant keys prevent a safe merge. - -## Root Guidance Ownership - -When root guidance is in scope, keep the target files in these roles: - -- `AGENTS.md`: strategic entrypoint, precedence anchor, naming contract, and runtime agent policy -- `LESSONS_LEARNED.md`: retained-learning ledger template aligned from source structure while preserving target-authored pending lessons; it remains non-canonical and repo-local in content -- `docs/README.md`: consumer-local routing guide for knowledge documents scaffolded only when missing and then preserved -- `docs/repository-context.md`: consumer-local descriptive context scaffolded only when missing and then preserved; it does not override policy -- `docs/architecture.md`: consumer-local architecture contract scaffolded only when missing and then preserved -- `docs/tech.md`: consumer-local technology contract scaffolded only when missing and then preserved -- `docs/structure.md`: consumer-local structure contract scaffolded only when missing and then preserved -- `.github/copilot-instructions.md`: review-only GitHub.com Copilot code review behavior -- `.github/INVENTORY.md`: exact live catalog generated from target filesystem state - -Do not flatten these roles into one file. Do not let target `AGENTS.md` become an inventory dump or a second full copy of review-only Copilot assets. - -## Tracking Plan Lifecycle - -- Write `tmp/copilot-sync.plan.md` before any mirrored change. -- Keep the plan in the target repository so the user can inspect pending sync work between runs. -- A follow-up `apply` against the same target therefore needs `--allow-dirty-target` when `tmp/copilot-sync.plan.md` is the only target diff left by `plan`; do not use that flag to bypass unrelated dirty target changes. -- When the target is dirty outside the mirrored sync scope, compare dirty paths against the planned managed mutations before deciding whether `--allow-dirty-target` is safe. -- If dirty target paths have zero overlap with planned managed mutations, treat them as plan-scoped manual reconciliation work and do not use `--allow-dirty-target` as a blanket bypass. -- When `apply` finishes and nothing remains pending, remove the plan file. -- If `apply` stops early or manual follow-up remains, keep the plan file for the next run. -- Remove legacy tracking artifacts named `internal-sync-*` from the target during `apply`. - -## Automation Entry Points - -- Preferred entrypoint: `./.github/scripts/run.sh sync_copilot_catalog` -- Audit entrypoint: `./.github/scripts/run.sh audit_copilot_catalog` -- Python entry point: `.github/scripts/sync_copilot_catalog.py` -- Audit entry point: `.github/scripts/audit_copilot_catalog.py` -- Core implementation: `.github/scripts/lib/syncing.py` -- Target manifest: `.github/copilot-sync.manifest.json` - -Prefer the shipped scripts when the request matches `plan`, `apply`, or a script-backed audit flow. Fall back to manual reasoning only when the request goes beyond the current automation contract. - -## Validation Sequence - -Use the closest existing checks for the touched behavior: - -1. `./.github/scripts/run.sh check_catalog_consistency --root . --include-token-risks` -2. `./.github/scripts/run.sh build_inventory --root .` -3. `./.github/scripts/run.sh sync_copilot_catalog plan --target-repo ` -4. `./.github/scripts/run.sh audit_copilot_catalog --root .` when the decision depends on governance drift or local override behavior -5. `pytest tests/test_sync_and_token_risks.py` when sync automation changes - -If a dedicated contract test is missing, call out the gap explicitly. - -When the target has no local catalog or contract validation script, confirm convergence from the source side with `python3 ./.github/scripts/sync_copilot_catalog.py plan --target-repo --format json` and require zero managed `create`, `update`, `ensure`, `rebuild`, or `delete` operations before treating the sync as converged. If that fallback check leaves `tmp/copilot-sync.plan.md` in an otherwise clean target and the file is not ignored, remove it after inspection. - -After `apply`, run the closest target-local catalog or contract validation when preserved `local-*` assets, preserved consumer-local GitHub instructions overrides, or other target-owned assets can still expose latent drift. Treat any resulting fixes as consumer-local follow-up work, not as source-baseline drift, unless the same finding reproduces against the source-managed assets themselves. - -## Reporting Contract - -Completed runs should make these facts visible: - -- target analysis and selected mode -- root-guidance alignment strategy and `LESSONS_LEARNED.md` status -- knowledge-document scaffold, preservation, legacy migration, and retired runtime document cleanup status -- preserved `local-*` assets status -- target-only cleanup decisions -- plan-file status and lifecycle -- validation results and remaining blockers - -End completed runs with `✅ Outcome`. -Keep `✅ Outcome` concise by default. -When additional provenance or execution detail would help, offer it as optional follow-up detail with a compact prompt that accepts number-only replies. -Include `🤖 Agents`, `📘 Instructions`, `📝 Prompts`, `🧩 Skills`, and `📦 Other Resources` only when those categories were actually used and the user asked for the detail, or when a narrower scoped contract requires inline disclosure. -State why each listed resource mattered in any included detail section. diff --git a/.github/skills/local-agent-sync-install-ai-resources/SKILL.md b/.github/skills/local-agent-sync-install-ai-resources/SKILL.md index 1a4d7573..c3edf60b 100644 --- a/.github/skills/local-agent-sync-install-ai-resources/SKILL.md +++ b/.github/skills/local-agent-sync-install-ai-resources/SKILL.md @@ -18,8 +18,10 @@ bundle directly. - Repository skill bundles are materialized only as absolute links under `~/.agents/skills/`. -- Root `AGENTS.md` is projected to `~/.agents/AGENTS.md` as a managed copy with - the complete `` block removed. +- Root `AGENTS.md` is projected to `~/.agents/AGENTS.md` as a managed copy. + Optional repository-local rules in `AGENTS.local.md` are not synchronized. +- Repository agents under `.github/agents/*.agent.md` are discovered + automatically except files whose name starts with `local-`. - Copilot agents are absolute links back to `.github/agents/`; Codex and OpenCode agents retain their translated copy paths. - Home-only skills are unmanaged and preserved. This includes catalog-excluded @@ -52,8 +54,9 @@ Accept `agents-md` as a CLI alias for the same target. - Keep `~/.agents/skills/` a real directory. Never replace the root with a link. - Treat repository root `AGENTS.md` as the only source of truth for the managed - `~/.agents/AGENTS.md` projection. Adopt and overwrite an unmanaged target - file, but never include `` in the result. + `~/.agents/AGENTS.md` projection. Treat `AGENTS.local.md` as an optional + repository-only layer that is never synchronized. Adopt and overwrite an + unmanaged target file. - Keep `~/.copilot/agents/` a real directory. Never replace the root with a link. - Create one canonical absolute link for every eligible repository skill. diff --git a/.github/skills/local-agent-sync-install-ai-resources/agents/openai.yaml b/.github/skills/local-agent-sync-install-ai-resources/agents/openai.yaml index 52c39581..676d5f1b 100644 --- a/.github/skills/local-agent-sync-install-ai-resources/agents/openai.yaml +++ b/.github/skills/local-agent-sync-install-ai-resources/agents/openai.yaml @@ -1,4 +1,4 @@ interface: display_name: "Home AI Resource Sync" short_description: "Sync AI resources and the portable AGENTS.md baseline to home" - default_prompt: "Use $local-agent-sync-install-ai-resources for repository-to-home sync. Treat an agents.md request as sync --targets agents.md, generating ~/.agents/AGENTS.md from root AGENTS.md without standards-repository-local-rules. Repository skills use managed absolute links; never sync home content into the repository. Default to compact output, require explicit approval for apply, and explain each blocker in plain language." + default_prompt: "Use $local-agent-sync-install-ai-resources for repository-to-home sync. Treat an agents.md request as sync --targets agents.md, generating ~/.agents/AGENTS.md from root AGENTS.md without the optional AGENTS.local.md policy. Repository skills use managed absolute links; never sync home content into the repository. Default to compact output, require explicit approval for apply, and explain each blocker in plain language." diff --git a/.github/skills/local-agent-sync-install-ai-resources/references/error-codes.md b/.github/skills/local-agent-sync-install-ai-resources/references/error-codes.md index 5ae2e8de..b7fb1c8a 100644 --- a/.github/skills/local-agent-sync-install-ai-resources/references/error-codes.md +++ b/.github/skills/local-agent-sync-install-ai-resources/references/error-codes.md @@ -21,7 +21,7 @@ code in an operator report. | `target-modified-managed` | A copied agent changed after the recorded hash. | Review the local change before replacing it. | | `source-missing` | A catalog source no longer exists. | Repair the catalog or source. | | `source-invalid-skill` | A repository skill lacks `SKILL.md`. | Repair the source bundle. | -| `source-invalid-agents-md` | Root `AGENTS.md` lacks the ordered shared and repository-local policy blocks required for safe projection. | Restore both blocks before updating `~/.agents/AGENTS.md`. | +| `source-invalid-agents-md` | Root `AGENTS.md` lacks the shared policy baseline required for safe projection. | Restore the shared baseline before updating `~/.agents/AGENTS.md`; `AGENTS.local.md` is optional and is not synchronized. | | `stale-managed` | A copied managed resource is no longer planned. | Review and use explicit `--prune-managed` if appropriate. | | `prune-not-approved` | Copied-resource pruning needs explicit approval. | Rerun apply with `--prune-managed`. | | `stale-content-drifted` | A stale copied resource changed locally. | Review it before deletion. | diff --git a/.github/skills/local-agent-sync-install-ai-resources/references/home-sync-catalog.yaml b/.github/skills/local-agent-sync-install-ai-resources/references/home-sync-catalog.yaml index d6f1ee28..f17b49c3 100644 --- a/.github/skills/local-agent-sync-install-ai-resources/references/home-sync-catalog.yaml +++ b/.github/skills/local-agent-sync-install-ai-resources/references/home-sync-catalog.yaml @@ -17,7 +17,7 @@ resources: include_targets: - agents.md target_support: See runtime support matrix - notes: Portable global baseline generated without standards-repository-local-rules. + notes: Portable global baseline generated without optional AGENTS.local.md policy. - resource_id: internal-gateway-idea source_family: agents source_path: .github/agents/internal-gateway-idea.agent.md @@ -27,18 +27,18 @@ resources: - opencode target_support: See runtime support matrix notes: Gateway agent for substantive idea definition and brainstorming. - - resource_id: internal-gateway-review + - resource_id: internal-gateway-review-generic source_family: agents - source_path: .github/agents/internal-gateway-review.agent.md + source_path: .github/agents/internal-gateway-review-generic.agent.md include_targets: - codex - copilot - opencode target_support: See runtime support matrix notes: Gateway agent for non-code and mixed review targets. - - resource_id: internal-review-code + - resource_id: internal-gateway-review-code source_family: agents - source_path: .github/agents/internal-review-code.agent.md + source_path: .github/agents/internal-gateway-review-code.agent.md include_targets: - codex - copilot diff --git a/.github/skills/local-agent-sync-install-ai-resources/references/runtime-support-matrix.yaml b/.github/skills/local-agent-sync-install-ai-resources/references/runtime-support-matrix.yaml index 1caeddfd..385533ef 100644 --- a/.github/skills/local-agent-sync-install-ai-resources/references/runtime-support-matrix.yaml +++ b/.github/skills/local-agent-sync-install-ai-resources/references/runtime-support-matrix.yaml @@ -9,7 +9,7 @@ rows: include_in_v1: true evidence: - Root AGENTS.md declares the shared baseline as the source for ~/.agents/AGENTS.md. - notes: Copy the portable root policy projection after removing standards-repository-local-rules. + notes: Copy the portable root policy projection without optional AGENTS.local.md policy. - target: skills resource_family: skills support_level: Documented diff --git a/.github/skills/local-agent-sync-install-ai-resources/references/sync-contract.md b/.github/skills/local-agent-sync-install-ai-resources/references/sync-contract.md index 80b16436..3e9ec81f 100644 --- a/.github/skills/local-agent-sync-install-ai-resources/references/sync-contract.md +++ b/.github/skills/local-agent-sync-install-ai-resources/references/sync-contract.md @@ -6,7 +6,8 @@ Use this reference for the exact repository-to-home contract. - `.github/skills/` is the sole source of truth for managed skill bundles. - Root `AGENTS.md` is the sole source of truth for the managed global - `~/.agents/AGENTS.md` baseline. + `~/.agents/AGENTS.md` baseline. A repository-root `AGENTS.local.md` is an + optional local policy layer and is not synchronized. - `~/.agents/skills/` remains a real directory. It is a runtime projection, not a second source. - Eligible skills are materialized as one absolute canonical symbolic link per @@ -16,9 +17,8 @@ Use this reference for the exact repository-to-home contract. - Never copy, merge, or reconcile home skill content into the repository. - Preserve all home-only skills, including `graphify`, every `local-*` bundle, invalid repository bundles, and every catalog-excluded ID. -- For the `agents.md` target, remove the complete - `` block and preserve the rest of root - `AGENTS.md`. Adopt and overwrite an unmanaged home copy. +- For the `agents.md` target, render and preserve root `AGENTS.md`. Do not read + or materialize `AGENTS.local.md`. Adopt and overwrite an unmanaged home copy. ## State And Manifest diff --git a/.github/skills/local-agent-sync-install-ai-resources/scripts/home_sync_contract.py b/.github/skills/local-agent-sync-install-ai-resources/scripts/home_sync_contract.py index 183c5865..22e38cf6 100644 --- a/.github/skills/local-agent-sync-install-ai-resources/scripts/home_sync_contract.py +++ b/.github/skills/local-agent-sync-install-ai-resources/scripts/home_sync_contract.py @@ -23,6 +23,7 @@ "copilot": Path(".copilot/agents"), "opencode": Path(".config/opencode/agents"), } +AGENT_TARGETS = ("codex", "copilot", "opencode") @dataclass(frozen=True) @@ -110,13 +111,28 @@ def load_home_sync_catalog(source_root: Path) -> list[CatalogResource]: if (resource["resource_id"], resource["source_family"]) not in explicit_ids ) + explicit_agent_ids = { + (resource.get("resource_id", ""), resource.get("source_family", "")) + for resource in resources + if isinstance(resource, dict) + } + resources.extend( + resource + for resource in discover_agent_resources(source_root) + if (resource["resource_id"], resource["source_family"]) not in explicit_agent_ids + ) + filtered = [] for resource in resources: rid = resource.get("resource_id", "") source_family = resource.get("source_family", "") if source_family == "skills" and rid in policy.excluded_skills: continue - if not policy.include_local_skills and rid.startswith("local-"): + if ( + source_family == "skills" + and not policy.include_local_skills + and rid.startswith("local-") + ): continue if not policy.include_internal_skills and rid.startswith("internal-"): continue @@ -192,6 +208,29 @@ def discover_skill_resources( return resources +def discover_agent_resources(source_root: Path) -> list[dict[str, object]]: + agents_root = source_root / ".github" / "agents" + if not agents_root.is_dir(): + return [] + + resources: list[dict[str, object]] = [] + for agent_path in sorted(agents_root.glob("*.agent.md")): + resource_id = agent_path.name.removesuffix(".agent.md") + if resource_id.startswith("local-"): + continue + resources.append( + { + "resource_id": resource_id, + "source_family": "agents", + "source_path": agent_path.relative_to(source_root).as_posix(), + "include_targets": list(AGENT_TARGETS), + "target_support": "See runtime support matrix", + "notes": "Auto-discovered agent.", + } + ) + return resources + + def resolve_skill_reference(source_root: Path, relative_path: Path) -> Path: source_candidate = source_root / SKILL_ROOT_RELATIVE / relative_path if source_candidate.exists(): diff --git a/.github/skills/local-agent-sync-install-ai-resources/scripts/home_syncing.py b/.github/skills/local-agent-sync-install-ai-resources/scripts/home_syncing.py index c540dad0..50ed2c21 100644 --- a/.github/skills/local-agent-sync-install-ai-resources/scripts/home_syncing.py +++ b/.github/skills/local-agent-sync-install-ai-resources/scripts/home_syncing.py @@ -44,10 +44,6 @@ TEXT_EXTENSIONS = (".md", ".txt", ".yml", ".yaml", ".json", ".sh", ".py") AGENTS_MD_FAMILY = "agents-md" AGENTS_MD_TARGET = "agents.md" -SHARED_BASELINE_START = "``" -SHARED_BASELINE_END = "``" -LOCAL_RULES_START = "``" -LOCAL_RULES_END = "``" @dataclass(frozen=True) @@ -826,7 +822,7 @@ def add_resource_blockers( "source-missing": "Catalog entry points to a source path that does not exist. Blocked to avoid materializing a stale or incomplete resource and to surface catalog drift.", "source-invalid-skill": "Source skill bundle is missing SKILL.md. Blocked because a valid repository skill bundle must contain SKILL.md.", "source-invalid-agent": "Source agent file is missing or not a .md file. Blocked because only allowlisted .agent.md files are eligible for linking or translation.", - "source-invalid-agents-md": "Root AGENTS.md is missing the portable shared baseline or the repository-local rules block. Blocked because the global projection must remove repository-only policy deterministically.", + "source-invalid-agents-md": "Root AGENTS.md is missing or unreadable. Blocked because the global projection needs a deterministic repository-wide policy source.", }[code] for target in intersection_targets(resource, targets): add_blocked_operation( @@ -1489,20 +1485,7 @@ def render_portable_agents_md(source_path: Path) -> str: if not source_path.is_file(): raise ValueError("source-invalid-agents-md: root AGENTS.md is missing") source = source_path.read_text(encoding="utf-8") - shared_start = source.find(SHARED_BASELINE_START) - shared_end = source.find(SHARED_BASELINE_END, shared_start + 1) - local_start = source.find(LOCAL_RULES_START) - local_end = source.find(LOCAL_RULES_END, local_start + 1) - if not (0 <= shared_start < shared_end < local_start < local_end): - raise ValueError( - "source-invalid-agents-md: expected ordered shared-baseline and standards-repository-local-rules blocks" - ) - - prefix = source[:local_start].rstrip() - suffix = source[local_end + len(LOCAL_RULES_END) :].strip() - if suffix: - return f"{prefix}\n\n{suffix}\n" - return f"{prefix}\n" + return f"{source.rstrip()}\n" def hash_portable_agents_md(source_path: Path) -> str: diff --git a/.github/skills/local-sync-repos/SKILL.md b/.github/skills/local-sync-repos/SKILL.md new file mode 100644 index 00000000..c4d65770 --- /dev/null +++ b/.github/skills/local-sync-repos/SKILL.md @@ -0,0 +1,58 @@ +--- +name: local-sync-repos +description: Use when aligning a consumer repository to this repository's managed instruction, root-policy, and shared-hygiene baseline while preserving target local-* assets. +--- + +# Local Sync Repos + +## Referenced skills + +- None. + +Use this skill as the mandatory operating engine for `.github/agents/local-sync-repos.agent.md`. + +## Scope + +Manage only these exact target paths: + +- `AGENTS.md` +- `.python-version` +- `.pre-commit-config.yaml` +- `.editorconfig` +- `.github/copilot-instructions.md` +- `.github/workflows/_pre-commit.yml` +- `.github/instructions/**` (source-authoritative; preserve target `local-*` filenames) +- `AGENTS.local.md` (create-once from template; never overwrite or delete) + +Do not synchronize agents, skills, prompts, inventory, documentation, lesson ledgers, VS Code settings, or unrelated workflows. + +## Commands + +```bash +python3 .github/skills/local-sync-repos/scripts/sync_repos.py plan --source-root . --target-repo --format compact +python3 .github/skills/local-sync-repos/scripts/sync_repos.py apply --source-root . --target-repo --format compact +``` + +- `plan` is the default safe mode. It writes only `tmp/local-sync-repos.plan.md` in the target. +- `apply` requires an existing matching plan fingerprint and blocks on dirty managed overlap or stale plans. + +## Safety Contract + +- Mirror source-managed files exactly. +- Preserve target `.github/instructions/**/local-*` files byte-identical. +- Delete target-only non-local instruction files only during an explicitly requested `apply`. +- Create `AGENTS.local.md` only when missing and never overwrite or delete an existing target copy. +- Block `apply` when a dirty target path overlaps a planned managed mutation. +- Block `apply` when the saved plan fingerprint does not match the current plan. +- Do not expose `--force`, `--allow-dirty-target`, or broader category selectors. +- Do not include commit or push steps; repository history remains user-owned. + +## Load On Demand + +- Read `references/sync-contract.md` for exact path ownership, action semantics, error codes, and convergence criteria. + +## Validation + +- Focused pytest: `python3.13 -m pytest tests/github/skills/local-sync-repos -q` +- Strict skill validation: `python3.13 ./.github/scripts/validate_internal_skills.py --skill local-sync-repos --strict` +- Catalog consistency: `python3.13 ./.github/scripts/check_catalog_consistency.py --root . --include-token-risks` diff --git a/.github/skills/local-sync-repos/agents/openai.yaml b/.github/skills/local-sync-repos/agents/openai.yaml new file mode 100644 index 00000000..3fa40bb1 --- /dev/null +++ b/.github/skills/local-sync-repos/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "Local Sync Repos" + short_description: "Mirror the managed instruction and hygiene baseline into consumer repos" + default_prompt: "Use $local-sync-repos to plan or apply a consumer-repository baseline sync from this standards repository." diff --git a/.github/skills/local-sync-repos/references/sync-contract.md b/.github/skills/local-sync-repos/references/sync-contract.md new file mode 100644 index 00000000..c8f3fd0a --- /dev/null +++ b/.github/skills/local-sync-repos/references/sync-contract.md @@ -0,0 +1,52 @@ +# Sync Contract Reference + +## Path Ownership + +### Source-managed exact copy + +These files are mirrored byte-exact from source to target: + +- `AGENTS.md` +- `.python-version` +- `.pre-commit-config.yaml` +- `.editorconfig` +- `.github/copilot-instructions.md` +- `.github/workflows/_pre-commit.yml` + +### Source-managed instructions + +Source files under `.github/instructions/` are mirrored to the target. The planner discovers them recursively. + +### Target-local instructions (preserved) + +Any file under `.github/instructions/` whose filename starts with `local-` is preserved byte-identical. The planner never mutates these paths. + +### Target-only non-local instructions (deleted on apply) + +Any file under `.github/instructions/` that is not in the source and does not start with `local-` is planned for deletion during `apply`. + +### AGENTS.local.md (create-once) + +Created from `templates/AGENTS.local.md` only when the target lacks the file. An existing target `AGENTS.local.md` is preserved byte-identical and never overwritten or deleted. + +## Action Semantics + +| Action | Meaning | +| --- | --- | +| `create` | File is missing in the target; will be written on apply. | +| `update` | File exists in the target but differs from source; will be overwritten on apply. | +| `delete` | Target-only non-local instruction; will be removed on apply. | +| `preserve` | File matches source or is consumer-owned; no mutation on apply. | + +## Error Codes + +| Code | Condition | +| --- | --- | +| `missing-plan` | Apply requested without a saved plan file. | +| `dirty-managed-overlap` | A dirty target path overlaps a planned managed mutation. | +| `stale-plan` | Saved plan fingerprint does not match the current plan. | +| `source-contract` | A required source file is missing or source and target resolve to the same directory. | + +## Convergence + +A target is converged when a fresh `plan` reports zero managed mutations (`managed_mutation_paths` is empty). After a converged apply, the target plan file is removed. diff --git a/.github/skills/local-sync-repos/scripts/sync_contract.py b/.github/skills/local-sync-repos/scripts/sync_contract.py new file mode 100644 index 00000000..8fdade73 --- /dev/null +++ b/.github/skills/local-sync-repos/scripts/sync_contract.py @@ -0,0 +1,284 @@ +from __future__ import annotations + +import hashlib +import json +import subprocess +from dataclasses import dataclass +from pathlib import Path +from typing import Literal + +MANAGED_COPY_PATHS: tuple[str, ...] = ( + "AGENTS.md", + ".python-version", + ".pre-commit-config.yaml", + ".editorconfig", + ".github/copilot-instructions.md", + ".github/workflows/_pre-commit.yml", +) + +_INSTRUCTION_ROOT = ".github/instructions" +_AGENTS_LOCAL = "AGENTS.local.md" +_TEMPLATE_RELATIVE = Path(__file__).resolve().parent.parent / "templates" / "AGENTS.local.md" + + +class SourceContractError(RuntimeError): + pass + + +@dataclass(frozen=True) +class Operation: + action: Literal["create", "update", "delete", "preserve"] + path: str + reason: str + source_sha256: str | None = None + target_sha256: str | None = None + + @property + def is_mutation(self) -> bool: + return self.action in {"create", "update", "delete"} + + +@dataclass(frozen=True) +class SyncPlan: + source_root: Path + target_root: Path + operations: tuple[Operation, ...] + dirty_managed_overlap: tuple[str, ...] + fingerprint: str + + @property + def managed_mutation_paths(self) -> tuple[str, ...]: + return tuple(item.path for item in self.operations if item.is_mutation) + + +def _sha256_bytes(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +def _sha256_path(path: Path) -> str: + return _sha256_bytes(path.read_bytes()) + + +def _normalize_posix(path: str) -> str: + return str(Path(path)).replace("\\", "/") + + +def dirty_paths(target_root: Path) -> frozenset[str]: + result = subprocess.run( + ["git", "-C", str(target_root), "status", "--porcelain=v1", "-z"], + check=True, + capture_output=True, + ) + raw = result.stdout + if not raw: + return frozenset() + paths: set[str] = set() + for entry in raw.split(b"\x00"): + if not entry: + continue + path_part = entry[3:].decode("utf-8", errors="replace") + if not path_part: + continue + paths.add(_normalize_posix(path_part)) + return frozenset(paths) + + +def _discover_source_instructions(source_root: Path) -> list[str]: + instruction_dir = source_root / _INSTRUCTION_ROOT + if not instruction_dir.is_dir(): + return [] + found: list[str] = [] + for path in sorted(instruction_dir.rglob("*")): + if path.is_file(): + found.append(_normalize_posix(path.relative_to(source_root).as_posix())) + return found + + +def _discover_target_instructions(target_root: Path) -> list[str]: + instruction_dir = target_root / _INSTRUCTION_ROOT + if not instruction_dir.is_dir(): + return [] + found: list[str] = [] + for path in sorted(instruction_dir.rglob("*")): + if path.is_file(): + found.append(_normalize_posix(path.relative_to(target_root).as_posix())) + return found + + +def _is_local_instruction(path: str) -> bool: + return Path(path).name.startswith("local-") + + +def build_plan(source_root: Path, target_root: Path) -> SyncPlan: + source = source_root.resolve() + target = target_root.resolve() + if source == target: + raise SourceContractError("source and target resolve to the same directory") + + for relative in MANAGED_COPY_PATHS: + candidate = source / relative + if not candidate.is_file(): + raise SourceContractError(f"missing required source path: {relative}") + + template_path = _TEMPLATE_RELATIVE + if not template_path.is_file(): + raise SourceContractError(f"missing AGENTS.local.md template: {template_path}") + + operations: list[Operation] = [] + + for relative in MANAGED_COPY_PATHS: + source_file = source / relative + target_file = target / relative + source_hash = _sha256_path(source_file) + if target_file.is_file(): + target_hash = _sha256_path(target_file) + if source_hash == target_hash: + operations.append( + Operation( + action="preserve", + path=_normalize_posix(relative), + reason="target matches source", + source_sha256=source_hash, + target_sha256=target_hash, + ) + ) + else: + operations.append( + Operation( + action="update", + path=_normalize_posix(relative), + reason="target differs from source", + source_sha256=source_hash, + target_sha256=target_hash, + ) + ) + else: + operations.append( + Operation( + action="create", + path=_normalize_posix(relative), + reason="missing in target", + source_sha256=source_hash, + ) + ) + + source_instructions = _discover_source_instructions(source) + target_instructions = _discover_target_instructions(target) + source_instruction_set = set(source_instructions) + target_instruction_set = set(target_instructions) + + for rel in sorted(source_instruction_set | target_instruction_set): + if _is_local_instruction(rel): + target_file = target / rel + target_hash = _sha256_path(target_file) if target_file.is_file() else None + operations.append( + Operation( + action="preserve", + path=_normalize_posix(rel), + reason="target-local instruction", + target_sha256=target_hash, + ) + ) + elif rel in source_instruction_set and rel in target_instruction_set: + source_hash = _sha256_path(source / rel) + target_hash = _sha256_path(target / rel) + if source_hash == target_hash: + operations.append( + Operation( + action="preserve", + path=_normalize_posix(rel), + reason="target matches source", + source_sha256=source_hash, + target_sha256=target_hash, + ) + ) + else: + operations.append( + Operation( + action="update", + path=_normalize_posix(rel), + reason="target differs from source", + source_sha256=source_hash, + target_sha256=target_hash, + ) + ) + elif rel in source_instruction_set: + source_hash = _sha256_path(source / rel) + operations.append( + Operation( + action="create", + path=_normalize_posix(rel), + reason="missing in target", + source_sha256=source_hash, + ) + ) + else: + target_hash = _sha256_path(target / rel) + operations.append( + Operation( + action="delete", + path=_normalize_posix(rel), + reason="target-only non-local instruction", + target_sha256=target_hash, + ) + ) + + agents_local_target = target / _AGENTS_LOCAL + if agents_local_target.is_file(): + target_hash = _sha256_path(agents_local_target) + operations.append( + Operation( + action="preserve", + path=_AGENTS_LOCAL, + reason="consumer-owned local policy", + target_sha256=target_hash, + ) + ) + else: + template_bytes = template_path.read_bytes() + source_hash = _sha256_bytes(template_bytes) + operations.append( + Operation( + action="create", + path=_AGENTS_LOCAL, + reason="create-once from template", + source_sha256=source_hash, + ) + ) + + operations.sort(key=lambda op: (op.path, op.action)) + + mutations = tuple(op for op in operations if op.is_mutation) + fingerprint = plan_fingerprint(mutations) + + target_dirty = dirty_paths(target) + mutation_paths = {op.path for op in operations if op.is_mutation} + overlap = tuple(sorted(target_dirty & mutation_paths)) + + return SyncPlan( + source_root=source, + target_root=target, + operations=tuple(operations), + dirty_managed_overlap=overlap, + fingerprint=fingerprint, + ) + + +def plan_fingerprint(operations: tuple[Operation, ...]) -> str: + records: list[dict[str, str | None]] = [] + for op in sorted(operations, key=lambda o: (o.path, o.action)): + records.append( + { + "action": op.action, + "path": _normalize_posix(op.path), + "source_sha256": op.source_sha256, + "target_sha256": op.target_sha256, + } + ) + canonical = json.dumps( + records, + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + ) + return _sha256_bytes(canonical.encode("utf-8")) diff --git a/.github/skills/local-sync-repos/scripts/sync_repos.py b/.github/skills/local-sync-repos/scripts/sync_repos.py new file mode 100644 index 00000000..ae09fa69 --- /dev/null +++ b/.github/skills/local-sync-repos/scripts/sync_repos.py @@ -0,0 +1,241 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import argparse +import json +import os +import re +import sys +import tempfile +from collections import Counter +from pathlib import Path + +SCRIPT_DIR = Path(__file__).resolve().parent +sys.path.insert(0, SCRIPT_DIR.as_posix()) + +from sync_contract import ( # noqa: E402 + Operation, + SourceContractError, + SyncPlan, + build_plan, +) + +_PLAN_RELATIVE = "tmp/local-sync-repos.plan.md" +_FINGERPRINT_RE = re.compile(r"^Plan fingerprint:\s+([0-9a-f]{64})\s*$", re.MULTILINE) + + +def parse_args(argv: list[str] | None = None) -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Plan or apply a local-sync-repos baseline sync." + ) + parser.add_argument("command", choices=["plan", "apply"]) + parser.add_argument("--source-root", required=True) + parser.add_argument("--target-repo", required=True) + parser.add_argument("--format", choices=["compact", "json", "text"], default="compact") + return parser.parse_args(argv) + + +def _normalize_posix(path: str) -> str: + return str(Path(path)).replace("\\", "/") + + +def _reject_unsafe_path(path: str) -> None: + if os.path.isabs(path): + raise ValueError(f"absolute operation path rejected: {path}") + normalized = _normalize_posix(path) + parts = Path(normalized).parts + if ".." in parts: + raise ValueError(f"parent-traversal operation path rejected: {path}") + + +def _plan_path_for(target_root: Path) -> Path: + return target_root / _PLAN_RELATIVE + + +def _write_plan_file(plan: SyncPlan, plan_file: Path) -> None: + lines: list[str] = [ + "# local-sync-repos plan", + "", + f"Plan fingerprint: {plan.fingerprint}", + f"Source: {plan.source_root.as_posix()}", + f"Target: {plan.target_root.as_posix()}", + "", + "## Operations", + "", + ] + for op in plan.operations: + lines.append(f"- {op.action:9s} {op.path} :: {op.reason}") + if plan.dirty_managed_overlap: + lines.append("") + lines.append("## Dirty managed overlap") + lines.append("") + for path in plan.dirty_managed_overlap: + lines.append(f"- {path}") + lines.append("") + content = "\n".join(lines) + plan_file.parent.mkdir(parents=True, exist_ok=True) + fd, tmp = tempfile.mkstemp(dir=str(plan_file.parent), suffix=".tmp") + try: + with os.fdopen(fd, "w", encoding="utf-8") as handle: + handle.write(content) + Path(tmp).replace(plan_file) + except BaseException: + if Path(tmp).exists(): + Path(tmp).unlink() + raise + + +def _read_saved_fingerprint(plan_file: Path) -> str: + text = plan_file.read_text(encoding="utf-8") + match = _FINGERPRINT_RE.search(text) + if not match: + raise ValueError(f"saved plan has no parseable fingerprint: {plan_file}") + return match.group(1) + + +def _apply_operation(plan: SyncPlan, operation: Operation) -> None: + _reject_unsafe_path(operation.path) + target_file = plan.target_root / operation.path + if operation.action == "create" or operation.action == "update": + if operation.source_sha256 is None: + raise ValueError(f"create/update missing source hash: {operation.path}") + source_file = plan.source_root / operation.path + if not source_file.is_file(): + template_path = SCRIPT_DIR.parent / "templates" / "AGENTS.local.md" + if operation.path == "AGENTS.local.md" and template_path.is_file(): + source_bytes = template_path.read_bytes() + else: + source_bytes = source_file.read_bytes() + else: + source_bytes = source_file.read_bytes() + target_file.parent.mkdir(parents=True, exist_ok=True) + fd, tmp = tempfile.mkstemp(dir=str(target_file.parent), suffix=".tmp") + try: + with os.fdopen(fd, "wb") as handle: + handle.write(source_bytes) + Path(tmp).replace(target_file) + except BaseException: + if Path(tmp).exists(): + Path(tmp).unlink() + raise + elif operation.action == "delete": + if not operation.path.startswith(".github/instructions/"): + raise ValueError(f"delete outside instructions: {operation.path}") + if target_file.is_file(): + target_file.unlink() + elif operation.action == "preserve": + pass + + +def _build_compact_payload(plan: SyncPlan, mode: str, status: str) -> dict[str, object]: + operation_counts = Counter(op.action for op in plan.operations) + return { + "mode": mode, + "status": status, + "target_repo": plan.target_root.as_posix(), + "fingerprint": plan.fingerprint, + "operation_counts": { + "total": len(plan.operations), + "by_action": dict(sorted(operation_counts.items())), + }, + "managed_mutation_paths": list(plan.managed_mutation_paths), + "dirty_managed_overlap": list(plan.dirty_managed_overlap), + } + + +def _build_json_payload(plan: SyncPlan, mode: str, status: str) -> dict[str, object]: + return { + "mode": mode, + "status": status, + "target_repo": plan.target_root.as_posix(), + "fingerprint": plan.fingerprint, + "source_root": plan.source_root.as_posix(), + "operations": [ + { + "action": op.action, + "path": op.path, + "reason": op.reason, + "source_sha256": op.source_sha256, + "target_sha256": op.target_sha256, + } + for op in plan.operations + ], + "managed_mutation_paths": list(plan.managed_mutation_paths), + "dirty_managed_overlap": list(plan.dirty_managed_overlap), + } + + +def _render_text(plan: SyncPlan, mode: str) -> None: + print(f"local-sync-repos {mode} for {plan.target_root.as_posix()}") + print(f"Plan fingerprint: {plan.fingerprint}") + for op in plan.operations: + print(f"- {op.action:9s} {op.path} :: {op.reason}") + + +def _emit(plan: SyncPlan, mode: str, status: str, fmt: str) -> None: + if fmt == "json": + print(json.dumps(_build_json_payload(plan, mode, status), indent=2, sort_keys=True)) + elif fmt == "compact": + print(json.dumps(_build_compact_payload(plan, mode, status), sort_keys=True)) + else: + _render_text(plan, mode) + + +def run_plan(source_root: Path, target_root: Path, fmt: str) -> int: + plan = build_plan(source_root, target_root) + plan_file = _plan_path_for(target_root) + _write_plan_file(plan, plan_file) + _emit(plan, "plan", "ok", fmt) + return 0 + + +def run_apply(source_root: Path, target_root: Path, fmt: str) -> int: + plan_file = _plan_path_for(target_root) + if not plan_file.is_file(): + print("error: missing-plan — run `plan` before `apply`.", file=sys.stderr) + return 1 + saved_fingerprint = _read_saved_fingerprint(plan_file) + plan = build_plan(source_root, target_root) + if plan.dirty_managed_overlap: + print( + f"error: dirty-managed-overlap — {', '.join(plan.dirty_managed_overlap)}", + file=sys.stderr, + ) + _emit(plan, "apply", "dirty-managed-overlap", fmt) + return 1 + if plan.fingerprint != saved_fingerprint: + print( + "error: stale-plan — saved plan fingerprint does not match current plan.", + file=sys.stderr, + ) + _emit(plan, "apply", "stale-plan", fmt) + return 1 + for op in plan.operations: + if op.is_mutation: + _apply_operation(plan, op) + updated = build_plan(source_root, target_root) + if not any(op.is_mutation for op in updated.operations): + if plan_file.exists(): + plan_file.unlink() + _emit(updated, "apply", "ok", fmt) + return 0 + + +def main(argv: list[str] | None = None) -> int: + args = parse_args(argv) + source = Path(args.source_root).resolve() + target = Path(args.target_repo).resolve() + try: + if args.command == "plan": + return run_plan(source, target, args.format) + return run_apply(source, target, args.format) + except SourceContractError as error: + print(f"error: source-contract — {error}", file=sys.stderr) + return 1 + except ValueError as error: + print(f"error: {error}", file=sys.stderr) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/.github/skills/local-sync-repos/templates/AGENTS.local.md b/.github/skills/local-sync-repos/templates/AGENTS.local.md new file mode 100644 index 00000000..cb3078d8 --- /dev/null +++ b/.github/skills/local-sync-repos/templates/AGENTS.local.md @@ -0,0 +1,7 @@ +# AGENTS.local.md - Repository-Local Policy + +This file owns repository-specific additions to the shared `AGENTS.md` baseline. + +- Keep rules specific to this repository. +- Do not duplicate the shared baseline. +- Prefer the nearest path-level owner when narrower guidance is required. diff --git a/.github/skills/mattpocock-code-review/SKILL.md b/.github/skills/mattpocock-code-review/SKILL.md new file mode 100644 index 00000000..0e9918c7 --- /dev/null +++ b/.github/skills/mattpocock-code-review/SKILL.md @@ -0,0 +1,89 @@ +--- +name: mattpocock-code-review +description: Review the changes since a fixed point (commit, branch, tag, or merge-base) along two axes — Standards (does the code follow this repo's documented coding standards?) and Spec (does the code match what the originating issue/PRD asked for?). Runs both reviews in parallel sub-agents and reports them side by side. Use when the user wants to review a branch, a PR, work-in-progress changes, or asks to "review since X". +--- + +Two-axis review of the diff between `HEAD` and a fixed point the user supplies: + +- **Standards** — does the code conform to this repo's documented coding standards? +- **Spec** — does the code faithfully implement the originating issue / PRD / spec? + +Both axes run as **parallel sub-agents** so they don't pollute each other's context, then this skill aggregates their findings. + +The issue tracker should have been provided to you — run `/mattpocock-setup-matt-pocock-skills` if `docs/agents/issue-tracker.md` is missing. + +## Process + +### 1. Pin the fixed point + +Whatever the user said is the fixed point — a commit SHA, branch name, tag, `main`, `HEAD~5`, etc. If they didn't specify one, ask for it. + +Capture the diff command once: `git diff ...HEAD` (three-dot, so the comparison is against the merge-base). Also note the list of commits via `git log ..HEAD --oneline`. + +Before going further, confirm the fixed point resolves (`git rev-parse `) and the diff is non-empty. A bad ref or empty diff should fail here — not inside two parallel sub-agents. + +### 2. Identify the spec source + +Look for the originating spec, in this order: + +1. Issue references in the commit messages (`#123`, `Closes #45`, GitLab `!67`, etc.) — fetch via the workflow in `docs/agents/issue-tracker.md`. +2. A path the user passed as an argument. +3. A PRD/spec file under `docs/`, `specs/`, or `.scratch/` matching the branch name or feature. +4. If nothing is found, ask the user where the spec is. If they say there isn't one, the **Spec** sub-agent will skip and report "no spec available". + +### 3. Identify the standards sources + +Anything in the repo that documents how code should be written, such as `CODING_STANDARDS.md` or `CONTRIBUTING.md`. + +On top of whatever the repo documents, the Standards axis always carries the **smell baseline** below — a fixed set of Fowler code smells (_Refactoring_, ch.3) that applies even when a repo documents nothing. Two rules bind it: + +- **The repo overrides.** A documented repo standard always wins; where it endorses something the baseline would flag, suppress the smell. +- **Always a judgement call.** Each smell is a labelled heuristic ("possible Feature Envy"), never a hard violation — and, like any standard here, skip anything tooling already enforces. + +Each smell reads *what it is* → *how to fix*; match it against the diff: + +- **Mysterious Name** — a function, variable, or type whose name doesn't reveal what it does or holds. → rename it; if no honest name comes, the design's murky. +- **Duplicated Code** — the same logic shape appears in more than one hunk or file in the change. → extract the shared shape, call it from both. +- **Feature Envy** — a method that reaches into another object's data more than its own. → move the method onto the data it envies. +- **Data Clumps** — the same few fields or params keep travelling together (a type wanting to be born). → bundle them into one type, pass that. +- **Primitive Obsession** — a primitive or string standing in for a domain concept that deserves its own type. → give the concept its own small type. +- **Repeated Switches** — the same `switch`/`if`-cascade on the same type recurs across the change. → replace with polymorphism, or one map both sites share. +- **Shotgun Surgery** — one logical change forces scattered edits across many files in the diff. → gather what changes together into one module. +- **Divergent Change** — one file or module is edited for several unrelated reasons. → split so each module changes for one reason. +- **Speculative Generality** — abstraction, parameters, or hooks added for needs the spec doesn't have. → delete it; inline back until a real need shows. +- **Message Chains** — long `a.b().c().d()` navigation the caller shouldn't depend on. → hide the walk behind one method on the first object. +- **Middle Man** — a class or function that mostly just delegates onward. → cut it, call the real target direct. +- **Refused Bequest** — a subclass or implementer that ignores or overrides most of what it inherits. → drop the inheritance, use composition. + +### 4. Spawn both sub-agents in parallel + +Send a single message with two `Agent` tool calls. Use the `general-purpose` subagent for both. + +**Standards sub-agent prompt** — include: + +- The full diff command and commit list. +- The list of standards-source files you found in step 3, **plus the smell baseline from step 3** pasted in full — the sub-agent has no other access to it. +- The brief: "Report — per file/hunk where relevant — (a) every place the diff violates a documented standard: cite the standard (file + the rule); and (b) any baseline smell you spot: name it and quote the hunk. Distinguish hard violations from judgement calls — documented-standard breaches can be hard, but baseline smells are always judgement calls, and a documented repo standard overrides the baseline. Skip anything tooling enforces. Under 400 words." + +**Spec sub-agent prompt** — include: + +- The diff command and commit list. +- The path or fetched contents of the spec. +- The brief: "Report: (a) requirements the spec asked for that are missing or partial; (b) behaviour in the diff that wasn't asked for (scope creep); (c) requirements that look implemented but where the implementation looks wrong. Quote the spec line for each finding. Under 400 words." + +If the spec is missing, skip the Spec sub-agent and note this in the final report. + +### 5. Aggregate + +Present the two reports under `## Standards` and `## Spec` headings, verbatim or lightly cleaned. Do **not** merge or rerank findings — the two axes are deliberately separate (see _Why two axes_). + +End with a one-line summary: total findings per axis, and the worst issue _within each axis_ (if any). Don't pick a single winner across axes — that's the reranking the separation exists to prevent. + +## Why two axes + +A change can pass one axis and fail the other: + +- Code that follows every standard but implements the wrong thing → **Standards pass, Spec fail.** +- Code that does exactly what the issue asked but breaks the project's conventions → **Spec pass, Standards fail.** + +Reporting them separately stops one axis from masking the other. diff --git a/.github/skills/mattpocock-code-review/agents/openai.yaml b/.github/skills/mattpocock-code-review/agents/openai.yaml new file mode 100644 index 00000000..9076774b --- /dev/null +++ b/.github/skills/mattpocock-code-review/agents/openai.yaml @@ -0,0 +1,3 @@ +interface: + display_name: "Code Review" + short_description: "Review a diff on standards and spec" diff --git a/.github/skills/mattpocock-codebase-design/DEEPENING.md b/.github/skills/mattpocock-codebase-design/DEEPENING.md new file mode 100644 index 00000000..3938457b --- /dev/null +++ b/.github/skills/mattpocock-codebase-design/DEEPENING.md @@ -0,0 +1,37 @@ +# Deepening + +How to deepen a cluster of shallow modules safely, given its dependencies. Assumes the vocabulary in [SKILL.md](SKILL.md) — **module**, **interface**, **seam**, **adapter**. + +## Dependency categories + +When assessing a candidate for deepening, classify its dependencies. The category determines how the deepened module is tested across its seam. + +### 1. In-process + +Pure computation, in-memory state, no I/O. Always deepenable — merge the modules and test through the new interface directly. No adapter needed. + +### 2. Local-substitutable + +Dependencies that have local test stand-ins (PGLite for Postgres, in-memory filesystem). Deepenable if the stand-in exists. The deepened module is tested with the stand-in running in the test suite. The seam is internal; no port at the module's external interface. + +### 3. Remote but owned (Ports & Adapters) + +Your own services across a network boundary (microservices, internal APIs). Define a **port** (interface) at the seam. The deep module owns the logic; the transport is injected as an **adapter**. Tests use an in-memory adapter. Production uses an HTTP/gRPC/queue adapter. + +Recommendation shape: *"Define a port at the seam, implement an HTTP adapter for production and an in-memory adapter for testing, so the logic sits in one deep module even though it's deployed across a network."* + +### 4. True external (Mock) + +Third-party services (Stripe, Twilio, etc.) you don't control. The deepened module takes the external dependency as an injected port; tests provide a mock adapter. + +## Seam discipline + +- **One adapter means a hypothetical seam. Two adapters means a real one.** Don't introduce a port unless at least two adapters are justified (typically production + test). A single-adapter seam is just indirection. +- **Internal seams vs external seams.** A deep module can have internal seams (private to its implementation, used by its own tests) as well as the external seam at its interface. Don't expose internal seams through the interface just because tests use them. + +## Testing strategy: replace, don't layer + +- Old unit tests on shallow modules become waste once tests at the deepened module's interface exist — delete them. +- Write new tests at the deepened module's interface. The **interface is the test surface**. +- Tests assert on observable outcomes through the interface, not internal state. +- Tests should survive internal refactors — they describe behaviour, not implementation. If a test has to change when the implementation changes, it's testing past the interface. diff --git a/.github/skills/mattpocock-codebase-design/DESIGN-IT-TWICE.md b/.github/skills/mattpocock-codebase-design/DESIGN-IT-TWICE.md new file mode 100644 index 00000000..49a7c42a --- /dev/null +++ b/.github/skills/mattpocock-codebase-design/DESIGN-IT-TWICE.md @@ -0,0 +1,44 @@ +# Design It Twice + +When the user wants to explore alternative interfaces for a chosen deepening candidate, use this parallel sub-agent pattern. Based on "Design It Twice" (Ousterhout) — your first idea is unlikely to be the best. + +Uses the vocabulary in [SKILL.md](SKILL.md) — **module**, **interface**, **seam**, **adapter**, **leverage**. + +## Process + +### 1. Frame the problem space + +Before spawning sub-agents, write a user-facing explanation of the problem space for the chosen candidate: + +- The constraints any new interface would need to satisfy +- The dependencies it would rely on, and which category they fall into (see [DEEPENING.md](DEEPENING.md)) +- A rough illustrative code sketch to ground the constraints — not a proposal, just a way to make the constraints concrete + +Show this to the user, then immediately proceed to Step 2. The user reads and thinks while the sub-agents work in parallel. + +### 2. Spawn sub-agents + +Spawn 3+ sub-agents in parallel using the Agent tool. Each must produce a **radically different** interface for the deepened module. + +Prompt each sub-agent with a separate technical brief (file paths, coupling details, dependency category from [DEEPENING.md](DEEPENING.md), what sits behind the seam). The brief is independent of the user-facing problem-space explanation in Step 1. Give each agent a different design constraint: + +- Agent 1: "Minimize the interface — aim for 1–3 entry points max. Maximise leverage per entry point." +- Agent 2: "Maximise flexibility — support many use cases and extension." +- Agent 3: "Optimise for the most common caller — make the default case trivial." +- Agent 4 (if applicable): "Design around ports & adapters for cross-seam dependencies." + +Include both [SKILL.md](SKILL.md) vocabulary and CONTEXT.md vocabulary in the brief so each sub-agent names things consistently with the architecture language and the project's domain language. + +Each sub-agent outputs: + +1. Interface (types, methods, params — plus invariants, ordering, error modes) +2. Usage example showing how callers use it +3. What the implementation hides behind the seam +4. Dependency strategy and adapters (see [DEEPENING.md](DEEPENING.md)) +5. Trade-offs — where leverage is high, where it's thin + +### 3. Present and compare + +Present designs sequentially so the user can absorb each one, then compare them in prose. Contrast by **depth** (leverage at the interface), **locality** (where change concentrates), and **seam placement**. + +After comparing, give your own recommendation: which design you think is strongest and why. If elements from different designs would combine well, propose a hybrid. Be opinionated — the user wants a strong read, not a menu. diff --git a/.github/skills/mattpocock-codebase-design/SKILL.md b/.github/skills/mattpocock-codebase-design/SKILL.md new file mode 100644 index 00000000..6b25bedb --- /dev/null +++ b/.github/skills/mattpocock-codebase-design/SKILL.md @@ -0,0 +1,114 @@ +--- +name: mattpocock-codebase-design +description: Shared vocabulary for designing deep modules. Use when the user wants to design or improve a module's interface, find deepening opportunities, decide where a seam goes, make code more testable or AI-navigable, or when another skill needs the deep-module vocabulary. +--- + +# Codebase Design + +Design **deep modules**: a lot of behaviour behind a small interface, placed at a clean seam, testable through that interface. Use this language and these principles wherever code is being designed or restructured. The aim is leverage for callers, locality for maintainers, and testability for everyone. + +## Glossary + +Use these terms exactly — don't substitute "component," "service," "API," or "boundary." Consistent language is the whole point. + +**Module** — anything with an interface and an implementation. Deliberately scale-agnostic: a function, class, package, or tier-spanning slice. _Avoid_: unit, component, service. + +**Interface** — everything a caller must know to use the module correctly: the type signature, but also invariants, ordering constraints, error modes, required configuration, and performance characteristics. _Avoid_: API, signature (too narrow — they refer only to the type-level surface). + +**Implementation** — what's inside a module, its body of code. Distinct from **Adapter**: a thing can be a small adapter with a large implementation (a Postgres repo) or a large adapter with a small implementation (an in-memory fake). Reach for "adapter" when the seam is the topic; "implementation" otherwise. + +**Depth** — leverage at the interface: the amount of behaviour a caller (or test) can exercise per unit of interface they have to learn. A module is **deep** when a large amount of behaviour sits behind a small interface, **shallow** when the interface is nearly as complex as the implementation. + +**Seam** _(Michael Feathers)_ — a place where you can alter behaviour without editing in that place; the *location* at which a module's interface lives. Where to put the seam is its own design decision, distinct from what goes behind it. _Avoid_: boundary (overloaded with DDD's bounded context). + +**Adapter** — a concrete thing that satisfies an interface at a seam. Describes *role* (what slot it fills), not substance (what's inside). + +**Leverage** — what callers get from depth: more capability per unit of interface they learn. One implementation pays back across N call sites and M tests. + +**Locality** — what maintainers get from depth: change, bugs, knowledge, and verification concentrate in one place rather than spreading across callers. Fix once, fixed everywhere. + +## Deep vs shallow + +**Deep module** = small interface + lots of implementation: + +``` +┌─────────────────────┐ +│ Small Interface │ ← Few methods, simple params +├─────────────────────┤ +│ │ +│ Deep Implementation│ ← Complex logic hidden +│ │ +└─────────────────────┘ +``` + +**Shallow module** = large interface + little implementation (avoid): + +``` +┌─────────────────────────────────┐ +│ Large Interface │ ← Many methods, complex params +├─────────────────────────────────┤ +│ Thin Implementation │ ← Just passes through +└─────────────────────────────────┘ +``` + +When designing an interface, ask: + +- Can I reduce the number of methods? +- Can I simplify the parameters? +- Can I hide more complexity inside? + +## Principles + +- **Depth is a property of the interface, not the implementation.** A deep module can be internally composed of small, mockable, swappable parts — they just aren't part of the interface. A module can have **internal seams** (private to its implementation, used by its own tests) as well as the **external seam** at its interface. +- **The deletion test.** Imagine deleting the module. If complexity vanishes, it was a pass-through. If complexity reappears across N callers, it was earning its keep. +- **The interface is the test surface.** Callers and tests cross the same seam. If you want to test *past* the interface, the module is probably the wrong shape. +- **One adapter means a hypothetical seam. Two adapters means a real one.** Don't introduce a seam unless something actually varies across it. + +## Designing for testability + +Good interfaces make testing natural: + +1. **Accept dependencies, don't create them.** + + ```typescript + // Testable + function processOrder(order, paymentGateway) {} + + // Hard to test + function processOrder(order) { + const gateway = new StripeGateway(); + } + ``` + +2. **Return results, don't produce side effects.** + + ```typescript + // Testable + function calculateDiscount(cart): Discount {} + + // Hard to test + function applyDiscount(cart): void { + cart.total -= discount; + } + ``` + +3. **Small surface area.** Fewer methods = fewer tests needed. Fewer params = simpler test setup. + +## Relationships + +- A **Module** has exactly one **Interface** (the surface it presents to callers and tests). +- **Depth** is a property of a **Module**, measured against its **Interface**. +- A **Seam** is where a **Module**'s **Interface** lives. +- An **Adapter** sits at a **Seam** and satisfies the **Interface**. +- **Depth** produces **Leverage** for callers and **Locality** for maintainers. + +## Rejected framings + +- **Depth as ratio of implementation-lines to interface-lines** (Ousterhout): rewards padding the implementation. We use depth-as-leverage instead. +- **"Interface" as the TypeScript `interface` keyword or a class's public methods**: too narrow — interface here includes every fact a caller must know. +- **"Boundary"**: overloaded with DDD's bounded context. Say **seam** or **interface**. + +## Going deeper + +- **Deepening a cluster given its dependencies** — see [DEEPENING.md](DEEPENING.md): dependency categories, seam discipline, and replace-don't-layer testing. +- **Exploring alternative interfaces** — see [DESIGN-IT-TWICE.md](DESIGN-IT-TWICE.md): spin up parallel sub-agents to design the interface several radically different ways, then compare on depth, locality, and seam placement. diff --git a/.github/skills/mattpocock-codebase-design/agents/openai.yaml b/.github/skills/mattpocock-codebase-design/agents/openai.yaml new file mode 100644 index 00000000..3180715e --- /dev/null +++ b/.github/skills/mattpocock-codebase-design/agents/openai.yaml @@ -0,0 +1,3 @@ +interface: + display_name: "Codebase Design" + short_description: "Vocabulary for deep-module design" diff --git a/.github/skills/mattpocock-domain-modeling/ADR-FORMAT.md b/.github/skills/mattpocock-domain-modeling/ADR-FORMAT.md new file mode 100644 index 00000000..da7e78ec --- /dev/null +++ b/.github/skills/mattpocock-domain-modeling/ADR-FORMAT.md @@ -0,0 +1,47 @@ +# ADR Format + +ADRs live in `docs/adr/` and use sequential numbering: `0001-slug.md`, `0002-slug.md`, etc. + +Create the `docs/adr/` directory lazily — only when the first ADR is needed. + +## Template + +```md +# {Short title of the decision} + +{1-3 sentences: what's the context, what did we decide, and why.} +``` + +That's it. An ADR can be a single paragraph. The value is in recording *that* a decision was made and *why* — not in filling out sections. + +## Optional sections + +Only include these when they add genuine value. Most ADRs won't need them. + +- **Status** frontmatter (`proposed | accepted | deprecated | superseded by ADR-NNNN`) — useful when decisions are revisited +- **Considered Options** — only when the rejected alternatives are worth remembering +- **Consequences** — only when non-obvious downstream effects need to be called out + +## Numbering + +Scan `docs/adr/` for the highest existing number and increment by one. + +## When to offer an ADR + +All three of these must be true: + +1. **Hard to reverse** — the cost of changing your mind later is meaningful +2. **Surprising without context** — a future reader will look at the code and wonder "why on earth did they do it this way?" +3. **The result of a real trade-off** — there were genuine alternatives and you picked one for specific reasons + +If a decision is easy to reverse, skip it — you'll just reverse it. If it's not surprising, nobody will wonder why. If there was no real alternative, there's nothing to record beyond "we did the obvious thing." + +### What qualifies + +- **Architectural shape.** "We're using a monorepo." "The write model is event-sourced, the read model is projected into Postgres." +- **Integration patterns between contexts.** "Ordering and Billing communicate via domain events, not synchronous HTTP." +- **Technology choices that carry lock-in.** Database, message bus, auth provider, deployment target. Not every library — just the ones that would take a quarter to swap out. +- **Boundary and scope decisions.** "Customer data is owned by the Customer context; other contexts reference it by ID only." The explicit no-s are as valuable as the yes-s. +- **Deliberate deviations from the obvious path.** "We're using manual SQL instead of an ORM because X." Anything where a reasonable reader would assume the opposite. These stop the next engineer from "fixing" something that was deliberate. +- **Constraints not visible in the code.** "We can't use AWS because of compliance requirements." "Response times must be under 200ms because of the partner API contract." +- **Rejected alternatives when the rejection is non-obvious.** If you considered GraphQL and picked REST for subtle reasons, record it — otherwise someone will suggest GraphQL again in six months. diff --git a/.github/skills/mattpocock-domain-modeling/CONTEXT-FORMAT.md b/.github/skills/mattpocock-domain-modeling/CONTEXT-FORMAT.md new file mode 100644 index 00000000..eaf2a185 --- /dev/null +++ b/.github/skills/mattpocock-domain-modeling/CONTEXT-FORMAT.md @@ -0,0 +1,60 @@ +# CONTEXT.md Format + +## Structure + +```md +# {Context Name} + +{One or two sentence description of what this context is and why it exists.} + +## Language + +**Order**: +{A one or two sentence description of the term} +_Avoid_: Purchase, transaction + +**Invoice**: +A request for payment sent to a customer after delivery. +_Avoid_: Bill, payment request + +**Customer**: +A person or organization that places orders. +_Avoid_: Client, buyer, account +``` + +## Rules + +- **Be opinionated.** When multiple words exist for the same concept, pick the best one and list the others under `_Avoid_`. +- **Keep definitions tight.** One or two sentences max. Define what it IS, not what it does. +- **Only include terms specific to this project's context.** General programming concepts (timeouts, error types, utility patterns) don't belong even if the project uses them extensively. Before adding a term, ask: is this a concept unique to this context, or a general programming concept? Only the former belongs. +- **Group terms under subheadings** when natural clusters emerge. If all terms belong to a single cohesive area, a flat list is fine. + +## Single vs multi-context repos + +**Single context (most repos):** One `CONTEXT.md` at the repo root. + +**Multiple contexts:** A `CONTEXT-MAP.md` at the repo root lists the contexts, where they live, and how they relate to each other: + +```md +# Context Map + +## Contexts + +- [Ordering](./src/ordering/CONTEXT.md) — receives and tracks customer orders +- [Billing](./src/billing/CONTEXT.md) — generates invoices and processes payments +- [Fulfillment](./src/fulfillment/CONTEXT.md) — manages warehouse picking and shipping + +## Relationships + +- **Ordering → Fulfillment**: Ordering emits `OrderPlaced` events; Fulfillment consumes them to start picking +- **Fulfillment → Billing**: Fulfillment emits `ShipmentDispatched` events; Billing consumes them to generate invoices +- **Ordering ↔ Billing**: Shared types for `CustomerId` and `Money` +``` + +The skill infers which structure applies: + +- If `CONTEXT-MAP.md` exists, read it to find contexts +- If only a root `CONTEXT.md` exists, single context +- If neither exists, create a root `CONTEXT.md` lazily when the first term is resolved + +When multiple contexts exist, infer which one the current topic relates to. If unclear, ask. diff --git a/.github/skills/mattpocock-domain-modeling/SKILL.md b/.github/skills/mattpocock-domain-modeling/SKILL.md new file mode 100644 index 00000000..4dc6a570 --- /dev/null +++ b/.github/skills/mattpocock-domain-modeling/SKILL.md @@ -0,0 +1,74 @@ +--- +name: mattpocock-domain-modeling +description: Build and sharpen a project's domain model. Use when the user wants to pin down domain terminology or a ubiquitous language, record an architectural decision, or when another skill needs to maintain the domain model. +--- + +# Domain Modeling + +Actively build and sharpen the project's domain model as you design. This is the *active* discipline — challenging terms, inventing edge-case scenarios, and writing the glossary and decisions down the moment they crystallise. (Merely *reading* `CONTEXT.md` for vocabulary is not this skill — that's a one-line habit any skill can do. This skill is for when you're changing the model, not just consuming it.) + +## File structure + +Most repos have a single context: + +``` +/ +├── CONTEXT.md +├── docs/ +│ └── adr/ +│ ├── 0001-event-sourced-orders.md +│ └── 0002-postgres-for-write-model.md +└── src/ +``` + +If a `CONTEXT-MAP.md` exists at the root, the repo has multiple contexts. The map points to where each one lives: + +``` +/ +├── CONTEXT-MAP.md +├── docs/ +│ └── adr/ ← system-wide decisions +├── src/ +│ ├── ordering/ +│ │ ├── CONTEXT.md +│ │ └── docs/adr/ ← context-specific decisions +│ └── billing/ +│ ├── CONTEXT.md +│ └── docs/adr/ +``` + +Create files lazily — only when you have something to write. If no `CONTEXT.md` exists, create one when the first term is resolved. If no `docs/adr/` exists, create it when the first ADR is needed. + +## During the session + +### Challenge against the glossary + +When the user uses a term that conflicts with the existing language in `CONTEXT.md`, call it out immediately. "Your glossary defines 'cancellation' as X, but you seem to mean Y — which is it?" + +### Sharpen fuzzy language + +When the user uses vague or overloaded terms, propose a precise canonical term. "You're saying 'account' — do you mean the Customer or the User? Those are different things." + +### Discuss concrete scenarios + +When domain relationships are being discussed, stress-test them with specific scenarios. Invent scenarios that probe edge cases and force the user to be precise about the boundaries between concepts. + +### Cross-reference with code + +When the user states how something works, check whether the code agrees. If you find a contradiction, surface it: "Your code cancels entire Orders, but you just said partial cancellation is possible — which is right?" + +### Update CONTEXT.md inline + +When a term is resolved, update `CONTEXT.md` right there. Don't batch these up — capture them as they happen. Use the format in [CONTEXT-FORMAT.md](./CONTEXT-FORMAT.md). + +`CONTEXT.md` should be totally devoid of implementation details. Do not treat `CONTEXT.md` as a spec, a scratch pad, or a repository for implementation decisions. It is a glossary and nothing else. + +### Offer ADRs sparingly + +Only offer to create an ADR when all three are true: + +1. **Hard to reverse** — the cost of changing your mind later is meaningful +2. **Surprising without context** — a future reader will wonder "why did they do it this way?" +3. **The result of a real trade-off** — there were genuine alternatives and you picked one for specific reasons + +If any of the three is missing, skip the ADR. Use the format in [ADR-FORMAT.md](./ADR-FORMAT.md). diff --git a/.github/skills/mattpocock-domain-modeling/agents/openai.yaml b/.github/skills/mattpocock-domain-modeling/agents/openai.yaml new file mode 100644 index 00000000..7f1522d2 --- /dev/null +++ b/.github/skills/mattpocock-domain-modeling/agents/openai.yaml @@ -0,0 +1,3 @@ +interface: + display_name: "Domain Modeling" + short_description: "Build and sharpen a domain model" diff --git a/.github/skills/mattpocock-grill-with-docs/SKILL.md b/.github/skills/mattpocock-grill-with-docs/SKILL.md new file mode 100644 index 00000000..58325e6f --- /dev/null +++ b/.github/skills/mattpocock-grill-with-docs/SKILL.md @@ -0,0 +1,7 @@ +--- +name: mattpocock-grill-with-docs +description: A relentless interview to sharpen a plan or design, which also creates docs (ADR's and glossary) as we go. +disable-model-invocation: true +--- + +Run a `/grill-me` session, using the `/mattpocock-domain-modeling` skill. diff --git a/.github/skills/mattpocock-grill-with-docs/agents/openai.yaml b/.github/skills/mattpocock-grill-with-docs/agents/openai.yaml new file mode 100644 index 00000000..5dbe2780 --- /dev/null +++ b/.github/skills/mattpocock-grill-with-docs/agents/openai.yaml @@ -0,0 +1,5 @@ +interface: + display_name: "Grill with Docs" + short_description: "Grill a design and write its docs" +policy: + allow_implicit_invocation: false diff --git a/.github/skills/mattpocock-handoff/agents/openai.yaml b/.github/skills/mattpocock-handoff/agents/openai.yaml new file mode 100644 index 00000000..6e1d8da1 --- /dev/null +++ b/.github/skills/mattpocock-handoff/agents/openai.yaml @@ -0,0 +1,5 @@ +interface: + display_name: "Handoff" + short_description: "Compact a conversation into a handoff" +policy: + allow_implicit_invocation: false diff --git a/.github/skills/mattpocock-implement/SKILL.md b/.github/skills/mattpocock-implement/SKILL.md new file mode 100644 index 00000000..40294b1c --- /dev/null +++ b/.github/skills/mattpocock-implement/SKILL.md @@ -0,0 +1,15 @@ +--- +name: mattpocock-implement +description: "Implement a piece of work based on a spec or set of tickets." +disable-model-invocation: true +--- + +Implement the work described by the user in the spec or tickets. + +Use /mattpocock-tdd where possible, at pre-agreed seams. + +Run typechecking regularly, single test files regularly, and the full test suite once at the end. + +Once done, use /mattpocock-code-review to review the work. + +Commit your work to the current branch. diff --git a/.github/skills/mattpocock-implement/agents/openai.yaml b/.github/skills/mattpocock-implement/agents/openai.yaml new file mode 100644 index 00000000..f8794dc1 --- /dev/null +++ b/.github/skills/mattpocock-implement/agents/openai.yaml @@ -0,0 +1,5 @@ +interface: + display_name: "Implement" + short_description: "Build work from a spec or tickets" +policy: + allow_implicit_invocation: false diff --git a/.github/skills/mattpocock-improve-codebase-architecture/HTML-REPORT.md b/.github/skills/mattpocock-improve-codebase-architecture/HTML-REPORT.md new file mode 100644 index 00000000..22fb4791 --- /dev/null +++ b/.github/skills/mattpocock-improve-codebase-architecture/HTML-REPORT.md @@ -0,0 +1,123 @@ +# HTML Report Format + +The architectural review is rendered as a single self-contained HTML file in the OS temp directory. Tailwind and Mermaid both come from CDNs. Mermaid handles graph-shaped diagrams reliably; hand-built divs and inline SVG handle the more editorial visuals (mass diagrams, cross-sections). Mix the two — don't lean on Mermaid for everything, it'll start to look generic. + +## Scaffold + +```html + + + + + Architecture review — {{repo name}} + + + + + +
+
...
+
...
+
...
+
+ + +``` + +## Header + +Repo name, date, and a compact legend: solid box = module, dashed line = seam, red arrow = leakage, thick dark box = deep module. No introduction paragraph — straight into the candidates. + +## Candidate card + +The diagrams carry the weight. Prose is sparse, plain, and uses the glossary terms (from the `/mattpocock-codebase-design` skill) without ceremony. + +Each candidate is one `
`: + +- **Title** — short, names the deepening (e.g. "Collapse the Order intake pipeline"). +- **Badge row** — recommendation strength (`Strong` = emerald, `Worth exploring` = amber, `Speculative` = slate), plus a tag for the dependency category (`in-process`, `local-substitutable`, `ports & adapters`, `mock`). +- **Files** — monospaced list, `font-mono text-sm`. +- **Before / After diagram** — the centrepiece. Two columns, side by side. See patterns below. +- **Problem** — one sentence. What hurts. +- **Solution** — one sentence. What changes. +- **Wins** — bullets, ≤6 words each. e.g. "Tests hit one interface", "Pricing logic stops leaking", "Delete 4 shallow wrappers". +- **ADR callout** (if applicable) — one line in an amber-tinted box. + +No paragraphs of explanation. If the diagram needs a paragraph to be understood, redraw the diagram. + +## Diagram patterns + +Pick the pattern that fits the candidate. Mix them. Don't make every diagram look the same — variety is part of the point. + +### Mermaid graph (the workhorse for dependencies / call flow) + +Use a Mermaid `flowchart` or `graph` when the point is "X calls Y calls Z, and look at the mess." Wrap it in a Tailwind-styled card so it doesn't feel parachuted in. Style with classDef to colour leakage edges red and the deep module dark. Sequence diagrams work well for "before: 6 round-trips; after: 1." + +```html +
+
+    flowchart LR
+      A[OrderHandler] --> B[OrderValidator]
+      B --> C[OrderRepo]
+      C -.leak.-> D[PricingClient]
+      classDef leak stroke:#dc2626,stroke-width:2px;
+      class C,D leak
+  
+
+``` + +### Hand-built boxes-and-arrows (when Mermaid's layout fights you) + +Modules as `
`s with borders and labels. Arrows as inline SVG `` or `` elements positioned absolutely over a relative container. Reach for this when you want the "after" diagram to feel like one thick-bordered deep module with greyed-out internals — Mermaid won't render that with the right weight. + +### Cross-section (good for layered shallowness) + +Stack horizontal bands (`h-12 border-l-4`) to show layers a call passes through. Before: 6 thin layers each doing nothing. After: 1 thick band labelled with the consolidated responsibility. + +### Mass diagram (good for "interface as wide as implementation") + +Two rectangles per module — one for interface surface area, one for implementation. Before: interface rectangle is nearly as tall as the implementation rectangle (shallow). After: interface rectangle is short, implementation rectangle is tall (deep). + +### Call-graph collapse + +Before: a tree of function calls rendered as nested boxes. After: the same tree collapsed into one box, with the now-internal calls shown faded inside it. + +## Style guidance + +- Lean editorial, not corporate-dashboard. Generous whitespace. Serif optional for headings (`font-serif` works well with stone/slate). +- Colour sparingly: one accent (emerald or indigo) plus red for leakage and amber for warnings. +- Keep diagrams ~320px tall so before/after sits comfortably side by side without scrolling. +- Use `text-xs uppercase tracking-wider` for module labels inside diagrams — they should read as schematic, not as UI. +- The only scripts are the Tailwind CDN and the Mermaid ESM import. The report is otherwise static — no app code, no interactivity beyond Mermaid's own rendering. + +## Top recommendation section + +One larger card. Candidate name, one sentence on why, anchor link to its card. That's it. + +## Tone + +Plain English, concise — but the architectural nouns and verbs come straight from the `/mattpocock-codebase-design` skill. Concision is not an excuse to drift. + +**Use exactly:** module, interface, implementation, depth, deep, shallow, seam, adapter, leverage, locality. + +**Never substitute:** component, service, unit (for module) · API, signature (for interface) · boundary (for seam) · layer, wrapper (for module, when you mean module). + +**Phrasings that fit the style:** + +- "Order intake module is shallow — interface nearly matches the implementation." +- "Pricing leaks across the seam." +- "Deepen: one interface, one place to test." +- "Two adapters justify the seam: HTTP in prod, in-memory in tests." + +**Wins bullets** name the gain in glossary terms: *"locality: bugs concentrate in one module"*, *"leverage: one interface, N call sites"*, *"interface shrinks; implementation absorbs the wrappers"*. Don't write *"easier to maintain"* or *"cleaner code"* — those terms aren't in the glossary and don't earn their place. + +No hedging, no throat-clearing, no "it's worth noting that…". If a sentence could be a bullet, make it a bullet. If a bullet could be cut, cut it. If a term isn't in the `/mattpocock-codebase-design` glossary, reach for one that is before inventing a new one. diff --git a/.github/skills/mattpocock-improve-codebase-architecture/SKILL.md b/.github/skills/mattpocock-improve-codebase-architecture/SKILL.md new file mode 100644 index 00000000..b728f40c --- /dev/null +++ b/.github/skills/mattpocock-improve-codebase-architecture/SKILL.md @@ -0,0 +1,70 @@ +--- +name: mattpocock-improve-codebase-architecture +description: Scan a codebase for deepening opportunities, present them as a visual HTML report, then grill through whichever one you pick. +--- + +# Improve Codebase Architecture + +Surface architectural friction and propose **deepening opportunities** — refactors that turn shallow modules into deep ones. The aim is testability and AI-navigability. + +This command is _informed_ by the project's domain model and built on a shared design vocabulary: + +- Run the `/mattpocock-codebase-design` skill for the architecture vocabulary (**module**, **interface**, **depth**, **seam**, **adapter**, **leverage**, **locality**) and its principles (the deletion test, "the interface is the test surface", "one adapter = hypothetical seam, two = real"). Use these terms exactly in every suggestion — don't drift into "component," "service," "API," or "boundary." +- The domain language in `CONTEXT.md` gives names to good seams; ADRs in `docs/adr/` record decisions this command should not re-litigate. + +## Process + +### 1. Explore + +**Scope before you scan — YAGNI.** Deepening a module pays off by making future changes to it easier, so put extra weight on the parts of the codebase that have recently changed. Decide *where* to look before you look: + +- If the user named a direction — a module, a subsystem, a pain point — take it, and skip the inference below. +- Otherwise, walk back a good stretch of the commit history (`git log --oneline`) to find the codebase's hot spots — the files and areas that keep coming up — and let those paths pull your attention first. If the changes are scattered with no clear hot spot, widen the net. + +Read the project's domain glossary (`CONTEXT.md`) and any ADRs in the area you're touching first. + +Then use the Agent tool with `subagent_type=Explore` to walk the codebase. Don't follow rigid heuristics — explore organically and note where you experience friction: + +- Where does understanding one concept require bouncing between many small modules? +- Where are modules **shallow** — interface nearly as complex as the implementation? +- Where have pure functions been extracted just for testability, but the real bugs hide in how they're called (no **locality**)? +- Where do tightly-coupled modules leak across their seams? +- Which parts of the codebase are untested, or hard to test through their current interface? + +Apply the **deletion test** to anything you suspect is shallow: would deleting it concentrate complexity, or just move it? A "yes, concentrates" is the signal you want. + +### 2. Present candidates as an HTML report + +Write a self-contained HTML file to the OS temp directory so nothing lands in the repo. Resolve the temp dir from `$TMPDIR`, falling back to `/tmp` (or `%TEMP%` on Windows), and write to `/architecture-review-.html` so each run gets a fresh file. Open it for the user — `xdg-open ` on Linux, `open ` on macOS, `start ` on Windows — and tell them the absolute path. + +The report uses **Tailwind via CDN** for layout and styling, and **Mermaid via CDN** for diagrams where a graph/flow/sequence reliably communicates the structure. Mix Mermaid with hand-crafted CSS/SVG visuals — use Mermaid when relationships are graph-shaped (call graphs, dependencies, sequences), and hand-built divs/SVG when you want something more editorial (mass diagrams, cross-sections, collapse animations). Each candidate gets a **before/after visualisation**. Be visual. + +For each candidate, render a card with: + +- **Files** — which files/modules are involved +- **Problem** — why the current architecture is causing friction +- **Solution** — plain English description of what would change +- **Benefits** — explained in terms of locality and leverage, and how tests would improve +- **Before / After diagram** — side-by-side, custom-drawn, illustrating the shallowness and the deepening +- **Recommendation strength** — one of `Strong`, `Worth exploring`, `Speculative`, rendered as a badge + +End the report with a **Top recommendation** section: which candidate you'd tackle first and why. + +**Use CONTEXT.md vocabulary for the domain, and the `/mattpocock-codebase-design` vocabulary for the architecture.** If `CONTEXT.md` defines "Order," talk about "the Order intake module" — not "the FooBarHandler," and not "the Order service." + +**ADR conflicts**: if a candidate contradicts an existing ADR, only surface it when the friction is real enough to warrant revisiting the ADR. Mark it clearly in the card (e.g. a warning callout: _"contradicts ADR-0007 — but worth reopening because…"_). Don't list every theoretical refactor an ADR forbids. + +See [HTML-REPORT.md](HTML-REPORT.md) for the full HTML scaffold, diagram patterns, and styling guidance. + +Do NOT propose interfaces yet. After the file is written, ask the user: "Which of these would you like to explore?" + +### 3. Grilling loop + +Once the user picks a candidate, run the `/grill-me` skill to walk the decision tree with them — constraints, dependencies, the shape of the deepened module, what sits behind the seam, what tests survive. + +Side effects happen inline as decisions crystallize — run the `/mattpocock-domain-modeling` skill to keep the domain model current as you go: + +- **Naming a deepened module after a concept not in `CONTEXT.md`?** Add the term to `CONTEXT.md`. Create the file lazily if it doesn't exist. +- **Sharpening a fuzzy term during the conversation?** Update `CONTEXT.md` right there. +- **User rejects the candidate with a load-bearing reason?** Offer an ADR, framed as: _"Want me to record this as an ADR so future architecture reviews don't re-suggest it?"_ Only offer when the reason would actually be needed by a future explorer to avoid re-suggesting the same thing — skip ephemeral reasons ("not worth it right now") and self-evident ones. +- **Want to explore alternative interfaces for the deepened module?** Run the `/mattpocock-codebase-design` skill and use its design-it-twice parallel sub-agent pattern. diff --git a/.github/skills/mattpocock-improve-codebase-architecture/agents/openai.yaml b/.github/skills/mattpocock-improve-codebase-architecture/agents/openai.yaml new file mode 100644 index 00000000..706fdca0 --- /dev/null +++ b/.github/skills/mattpocock-improve-codebase-architecture/agents/openai.yaml @@ -0,0 +1,5 @@ +interface: + display_name: "Improve Codebase Architecture" + short_description: "Find and grill architecture improvements" +policy: + allow_implicit_invocation: false diff --git a/.github/skills/mattpocock-research/agents/openai.yaml b/.github/skills/mattpocock-research/agents/openai.yaml new file mode 100644 index 00000000..e18b96ca --- /dev/null +++ b/.github/skills/mattpocock-research/agents/openai.yaml @@ -0,0 +1,3 @@ +interface: + display_name: "Research" + short_description: "Research from high-trust sources" diff --git a/.github/skills/mattpocock-setup-matt-pocock-skills/SKILL.md b/.github/skills/mattpocock-setup-matt-pocock-skills/SKILL.md new file mode 100644 index 00000000..b7498c65 --- /dev/null +++ b/.github/skills/mattpocock-setup-matt-pocock-skills/SKILL.md @@ -0,0 +1,116 @@ +--- +name: mattpocock-setup-matt-pocock-skills +description: Configure this repo for the engineering skills — set up its issue tracker, triage label vocabulary, and domain doc layout. Run once before first use of the other engineering skills. +disable-model-invocation: true +--- + +# Setup Matt Pocock's Skills + +Scaffold the per-repo configuration that the engineering skills assume: + +- **Issue tracker** — where issues live (GitHub by default; local markdown is also supported out of the box) +- **Triage labels** — the strings used for the five canonical triage roles +- **Domain docs** — where `CONTEXT.md` and ADRs live, and the consumer rules for reading them + +This is a prompt-driven skill, not a deterministic script. Explore, present what you found, confirm with the user, then write. + +## Process + +### 1. Explore + +Look at the current repo to understand its starting state. Read whatever exists; don't assume: + +- `git remote -v` and `.git/config` — is this a GitHub repo? Which one? +- `AGENTS.md` and `CLAUDE.md` at the repo root — does either exist? Is there already an `## Agent skills` section in either? +- `CONTEXT.md` and `CONTEXT-MAP.md` at the repo root +- `docs/adr/` and any `src/*/docs/adr/` directories +- `docs/agents/` — does this skill's prior output already exist? +- `.scratch/` — sign that a local-markdown issue tracker convention is already in use +- Is the `triage` skill installed? (a `triage` skill folder alongside this one, or `triage` in your available skills.) This decides whether Section B runs at all. +- Monorepo signals — a `pnpm-workspace.yaml`, a `workspaces` field in `package.json`, or a populated `packages/*` with its own `src/`. Present only in a genuinely large multi-package repo; their absence means single-context, which is almost every repo. + +### 2. Present findings and ask + +Summarise what's present and what's missing. Then take the sections in order — one section, one answer, then the next. + +Lead each section with the recommended answer so the user can accept it in a word. Give a one-line explainer only when the choice genuinely branches; skip the section entirely when exploration already settled it (Section B when `triage` isn't installed, Section C when there's no monorepo). + +**Section A — Issue tracker.** + +> Explainer: The "issue tracker" is where issues live for this repo. Skills like `to-tickets`, `triage`, `mattpocock-to-spec`, and `qa` read from and write to it — they need to know whether to call `gh issue create`, write a markdown file under `.scratch/`, or follow some other workflow you describe. Pick the place you actually track work for this repo. + +Default posture: these skills were designed for GitHub. If a `git remote` points at GitHub, propose that. If a `git remote` points at GitLab (`gitlab.com` or a self-hosted host), propose GitLab. Otherwise (or if the user prefers), offer: + +- **GitHub** — issues live in the repo's GitHub Issues (uses the `gh` CLI) +- **GitLab** — issues live in the repo's GitLab Issues (uses the [`glab`](https://gitlab.com/gitlab-org/cli) CLI) +- **Local markdown** — issues live as files under `.scratch//` in this repo (good for solo projects or repos without a remote) +- **Other** (Jira, Linear, etc.) — ask the user to describe the workflow in one paragraph; the skill will record it as freeform prose + +Record the choice in `docs/agents/issue-tracker.md`. The GitHub and GitLab templates carry a "PRs as a request surface" flag, defaulted **off** — leave it off and don't raise it; a user who wants external PRs in the triage queue can flip the flag in the file later. + +**Section B — Triage label vocabulary.** Skip this section entirely if the `triage` skill isn't installed (exploration told you) — an uninstalled skill needs no labels. + +If it is installed, ask exactly one question: + +> Do you want to keep the default triage labels? (recommended: **yes**) + +The defaults are the five canonical roles, each label string equal to its name: `needs-triage`, `needs-info`, `ready-for-agent`, `ready-for-human`, `wontfix`. On **yes**, write them as-is. Only if the user says no — usually because their tracker already uses other names (e.g. `bug:triage` for `needs-triage`) — collect the overrides so `triage` applies existing labels instead of creating duplicates. + +**Section C — Domain docs.** Default to **single-context** — one `CONTEXT.md` + `docs/adr/` at the repo root. This fits almost every repo; write it without asking. + +Offer **multi-context** — a root `CONTEXT-MAP.md` pointing to per-context `CONTEXT.md` files — only when exploration found monorepo signals. Then confirm which layout they want. + +### 3. Confirm and edit + +Show the user a draft of: + +- The `## Agent skills` block to add to whichever of `CLAUDE.md` / `AGENTS.md` is being edited (see step 4 for selection rules) +- The contents of `docs/agents/issue-tracker.md`, `docs/agents/domain.md`, and `docs/agents/triage-labels.md` (the last only when `triage` is installed) + +Let them edit before writing. + +### 4. Write + +**Pick the file to edit:** + +- If `CLAUDE.md` exists, edit it. +- Else if `AGENTS.md` exists, edit it. +- If neither exists, ask the user which one to create — don't pick for them. + +Never create `AGENTS.md` when `CLAUDE.md` already exists (or vice versa) — always edit the one that's already there. + +If an `## Agent skills` block already exists in the chosen file, update its contents in-place rather than appending a duplicate. Don't overwrite user edits to the surrounding sections. + +The block: + +```markdown +## Agent skills + +### Issue tracker + +[one-line summary of where issues are tracked]. See `docs/agents/issue-tracker.md`. + +### Triage labels + +[one-line summary of the label vocabulary]. See `docs/agents/triage-labels.md`. + +### Domain docs + +[one-line summary of layout — "single-context" or "multi-context"]. See `docs/agents/domain.md`. +``` + +Include the `### Triage labels` sub-block, and write `docs/agents/triage-labels.md`, only when `triage` is installed and Section B ran. When it isn't, both are omitted. + +Then write the docs files using the seed templates in this skill folder as a starting point: + +- [issue-tracker-github.md](./issue-tracker-github.md) — GitHub issue tracker +- [issue-tracker-gitlab.md](./issue-tracker-gitlab.md) — GitLab issue tracker +- [issue-tracker-local.md](./issue-tracker-local.md) — local-markdown issue tracker +- [triage-labels.md](./triage-labels.md) — label mapping (only if `triage` is installed) +- [domain.md](./domain.md) — domain doc consumer rules + layout + +For "other" issue trackers, write `docs/agents/issue-tracker.md` from scratch using the user's description. + +### 5. Done + +Tell the user the setup is complete and which engineering skills will now read from these files. Mention they can edit `docs/agents/*.md` directly later — re-running this skill is only necessary if they want to switch issue trackers or restart from scratch. diff --git a/.github/skills/mattpocock-setup-matt-pocock-skills/agents/openai.yaml b/.github/skills/mattpocock-setup-matt-pocock-skills/agents/openai.yaml new file mode 100644 index 00000000..65a0da81 --- /dev/null +++ b/.github/skills/mattpocock-setup-matt-pocock-skills/agents/openai.yaml @@ -0,0 +1,5 @@ +interface: + display_name: "Setup Matt Pocock Skills" + short_description: "Configure a repo for the skills" +policy: + allow_implicit_invocation: false diff --git a/.github/skills/mattpocock-setup-matt-pocock-skills/domain.md b/.github/skills/mattpocock-setup-matt-pocock-skills/domain.md new file mode 100644 index 00000000..a03afcc7 --- /dev/null +++ b/.github/skills/mattpocock-setup-matt-pocock-skills/domain.md @@ -0,0 +1,51 @@ +# Domain Docs + +How the engineering skills should consume this repo's domain documentation when exploring the codebase. + +## Before exploring, read these + +- **`CONTEXT.md`** at the repo root, or +- **`CONTEXT-MAP.md`** at the repo root if it exists — it points at one `CONTEXT.md` per context. Read each one relevant to the topic. +- **`docs/adr/`** — read ADRs that touch the area you're about to work in. In multi-context repos, also check `src//docs/adr/` for context-scoped decisions. + +If any of these files don't exist, **proceed silently**. Don't flag their absence; don't suggest creating them upfront. The `/mattpocock-domain-modeling` skill (reached via `/mattpocock-grill-with-docs` and `/improve-codebase-architecture`) creates them lazily when terms or decisions actually get resolved. + +## File structure + +Single-context repo (most repos): + +``` +/ +├── CONTEXT.md +├── docs/adr/ +│ ├── 0001-event-sourced-orders.md +│ └── 0002-postgres-for-write-model.md +└── src/ +``` + +Multi-context repo (presence of `CONTEXT-MAP.md` at the root): + +``` +/ +├── CONTEXT-MAP.md +├── docs/adr/ ← system-wide decisions +└── src/ + ├── ordering/ + │ ├── CONTEXT.md + │ └── docs/adr/ ← context-specific decisions + └── billing/ + ├── CONTEXT.md + └── docs/adr/ +``` + +## Use the glossary's vocabulary + +When your output names a domain concept (in an issue title, a refactor proposal, a hypothesis, a test name), use the term as defined in `CONTEXT.md`. Don't drift to synonyms the glossary explicitly avoids. + +If the concept you need isn't in the glossary yet, that's a signal — either you're inventing language the project doesn't use (reconsider) or there's a real gap (note it for `/mattpocock-domain-modeling`). + +## Flag ADR conflicts + +If your output contradicts an existing ADR, surface it explicitly rather than silently overriding: + +> _Contradicts ADR-0007 (event-sourced orders) — but worth reopening because…_ diff --git a/.github/skills/mattpocock-setup-matt-pocock-skills/issue-tracker-github.md b/.github/skills/mattpocock-setup-matt-pocock-skills/issue-tracker-github.md new file mode 100644 index 00000000..c4a3be7b --- /dev/null +++ b/.github/skills/mattpocock-setup-matt-pocock-skills/issue-tracker-github.md @@ -0,0 +1,45 @@ +# Issue tracker: GitHub + +Issues and PRDs for this repo live as GitHub issues. Use the `gh` CLI for all operations. + +## Conventions + +- **Create an issue**: `gh issue create --title "..." --body "..."`. Use a heredoc for multi-line bodies. +- **Read an issue**: `gh issue view --comments`, filtering comments by `jq` and also fetching labels. +- **List issues**: `gh issue list --state open --json number,title,body,labels,comments --jq '[.[] | {number, title, body, labels: [.labels[].name], comments: [.comments[].body]}]'` with appropriate `--label` and `--state` filters. +- **Comment on an issue**: `gh issue comment --body "..."` +- **Apply / remove labels**: `gh issue edit --add-label "..."` / `--remove-label "..."` +- **Close**: `gh issue close --comment "..."` + +Infer the repo from `git remote -v` — `gh` does this automatically when run inside a clone. + +## Pull requests as a triage surface + +**PRs as a request surface: no.** _(Set to `yes` if this repo treats external PRs as feature requests; `/triage` reads this flag.)_ + +When set to `yes`, PRs run through the same labels and states as issues, using the `gh pr` equivalents: + +- **Read a PR**: `gh pr view --comments` and `gh pr diff ` for the diff. +- **List external PRs for triage**: `gh pr list --state open --json number,title,body,labels,author,authorAssociation,comments` then keep only `authorAssociation` of `CONTRIBUTOR`, `FIRST_TIME_CONTRIBUTOR`, or `NONE` (drop `OWNER`/`MEMBER`/`COLLABORATOR`). +- **Comment / label / close**: `gh pr comment`, `gh pr edit --add-label`/`--remove-label`, `gh pr close`. + +GitHub shares one number space across issues and PRs, so a bare `#42` may be either — resolve with `gh pr view 42` and fall back to `gh issue view 42`. + +## When a skill says "publish to the issue tracker" + +Create a GitHub issue. + +## When a skill says "fetch the relevant ticket" + +Run `gh issue view --comments`. + +## Wayfinding operations + +Used by `/mattpocock-wayfinder`. The **map** is a single issue with **child** issues as tickets. + +- **Map**: a single issue labelled `wayfinder:map`, holding the Notes / Decisions-so-far / Fog body. `gh issue create --label wayfinder:map`. +- **Child ticket**: an issue linked to the map as a GitHub sub-issue (`gh api` on the sub-issues endpoint). Where sub-issues aren't enabled, add the child to a task list in the map body and put `Part of #` at the top of the child body. Labels: `wayfinder:` (`mattpocock-research`/`prototype`/`grilling`/`task`). Once claimed, the ticket is assigned to the driving dev. +- **Blocking**: GitHub's **native issue dependencies** — the canonical, UI-visible representation. Add an edge with `gh api --method POST repos///issues//dependencies/blocked_by -F issue_id=`, where `` is the blocker's numeric **database id** (`gh api repos///issues/ --jq .id`, _not_ the `#number` or `node_id`). GitHub reports `issue_dependencies_summary.blocked_by` (open blockers only — the live gate). Where dependencies aren't available, fall back to a `Blocked by: #, #` line at the top of the child body. A ticket is unblocked when every blocker is closed. +- **Frontier query**: list the map's open children (`gh issue list --state open`, scoped to the map's sub-issues / task list), drop any with an open blocker (`issue_dependencies_summary.blocked_by > 0`, or an open issue in the `Blocked by` line) or an assignee; first in map order wins. +- **Claim**: `gh issue edit --add-assignee @me` — the session's first write. +- **Resolve**: `gh issue comment --body ""`, then `gh issue close `, then append a context pointer (gist + link) to the map's Decisions-so-far. diff --git a/.github/skills/mattpocock-setup-matt-pocock-skills/issue-tracker-gitlab.md b/.github/skills/mattpocock-setup-matt-pocock-skills/issue-tracker-gitlab.md new file mode 100644 index 00000000..0c8c92b2 --- /dev/null +++ b/.github/skills/mattpocock-setup-matt-pocock-skills/issue-tracker-gitlab.md @@ -0,0 +1,46 @@ +# Issue tracker: GitLab + +Issues and PRDs for this repo live as GitLab issues. Use the [`glab`](https://gitlab.com/gitlab-org/cli) CLI for all operations. + +## Conventions + +- **Create an issue**: `glab issue create --title "..." --description "..."`. Use a heredoc for multi-line descriptions. Pass `--description -` to open an editor. +- **Read an issue**: `glab issue view --comments`. Use `-F json` for machine-readable output. +- **List issues**: `glab issue list -F json` with appropriate `--label` filters. +- **Comment on an issue**: `glab issue note --message "..."`. GitLab calls comments "notes". +- **Apply / remove labels**: `glab issue update --label "..."` / `--unlabel "..."`. Multiple labels can be comma-separated or by repeating the flag. +- **Close**: `glab issue close `. `glab issue close` does not accept a closing comment, so post the explanation first with `glab issue note --message "..."`, then close. +- **Merge requests**: GitLab calls PRs "merge requests". Use `glab mr create`, `glab mr view`, `glab mr note`, etc. — the same shape as `gh pr ...` with `mr` in place of `pr` and `note`/`--message` in place of `comment`/`--body`. + +Infer the repo from `git remote -v` — `glab` does this automatically when run inside a clone. + +## Merge requests as a triage surface + +**MRs as a request surface: no.** _(Set to `yes` if this repo treats external merge requests as feature requests; `/triage` reads this flag.)_ + +When set to `yes`, MRs run through the same labels and states as issues, using the `glab mr` equivalents: + +- **Read an MR**: `glab mr view --comments` and `glab mr diff ` for the diff. +- **List external MRs for triage**: `glab mr list -F json`, then keep only MRs whose author is not a project member/owner (a contributor's MR, not a maintainer's in-flight work). +- **Comment / label / close**: `glab mr note`, `glab mr update --label`/`--unlabel`, `glab mr close`. + +Unlike GitHub, GitLab numbers issues and MRs separately, so `#42` is unambiguous once you know which surface the maintainer means. + +## When a skill says "publish to the issue tracker" + +Create a GitLab issue. + +## When a skill says "fetch the relevant ticket" + +Run `glab issue view --comments`. + +## Wayfinding operations + +Used by `/mattpocock-wayfinder`. The **map** is a single issue with **child** issues as tickets. + +- **Map**: a single issue labelled `wayfinder:map`, holding the Notes / Decisions-so-far / Fog body. `glab issue create --label wayfinder:map`. (On GitLab tiers with native epics, an epic may hold the map instead; a labelled issue works everywhere.) +- **Child ticket**: an issue carrying `Part of #` at the top of its description and labels `wayfinder:` (`mattpocock-research`/`prototype`/`grilling`/`task`). Once claimed, the ticket is assigned to the driving dev. +- **Blocking**: GitLab's **native blocking link** — the canonical, UI-visible representation. Add it with the `/blocked_by #` quick action, posted as a note (`glab issue note --message "/blocked_by #"`). Native blocking links are a Premium/Ultimate feature; on the free tier (or where unavailable) fall back to a `Blocked by: #, #` line at the top of the description. A ticket is unblocked when every blocker is closed. +- **Frontier query**: `glab issue list -F json` scoped to the map's children, drop any with an open blocker — a native `blocked_by` link to an open issue (`glab api projects/:id/issues/:iid/links`), or an open issue in the `Blocked by` line — or an assignee; first in map order wins. +- **Claim**: `glab issue update --assignee @me` — the session's first write. +- **Resolve**: `glab issue note --message ""`, then `glab issue close `, then append a context pointer (gist + link) to the map's Decisions-so-far. diff --git a/.github/skills/mattpocock-setup-matt-pocock-skills/issue-tracker-local.md b/.github/skills/mattpocock-setup-matt-pocock-skills/issue-tracker-local.md new file mode 100644 index 00000000..a7f1a765 --- /dev/null +++ b/.github/skills/mattpocock-setup-matt-pocock-skills/issue-tracker-local.md @@ -0,0 +1,30 @@ +# Issue tracker: Local Markdown + +Issues and specs (you may know a spec as a PRD) for this repo live as markdown files in `.scratch/`. + +## Conventions + +- One feature per directory: `.scratch//` +- The spec is `.scratch//spec.md` +- Implementation issues are one file per ticket at `.scratch//issues/-.md`, numbered from `01` — never a single combined tickets file +- Triage state is recorded as a `Status:` line near the top of each issue file (see `triage-labels.md` for the role strings) +- Comments and conversation history append to the bottom of the file under a `## Comments` heading + +## When a skill says "publish to the issue tracker" + +Create a new file under `.scratch//` (creating the directory if needed). + +## When a skill says "fetch the relevant ticket" + +Read the file at the referenced path. The user will normally pass the path or the issue number directly. + +## Wayfinding operations + +Used by `/mattpocock-wayfinder`. The **map** is a file with one **child** file per ticket. + +- **Map**: `.scratch//map.md` — the Notes / Decisions-so-far / Fog body. +- **Child ticket**: `.scratch//issues/NN-.md`, numbered from `01`, with the question in the body. A `Type:` line records the ticket type (`mattpocock-research`/`prototype`/`grilling`/`task`); a `Status:` line records `claimed`/`resolved`. +- **Blocking**: a `Blocked by: NN, NN` line near the top. A ticket is unblocked when every file it lists is `resolved`. +- **Frontier**: scan `.scratch//issues/` for files that are open, unblocked, and unclaimed; first by number wins. +- **Claim**: set `Status: claimed` and save before any work. +- **Resolve**: append the answer under an `## Answer` heading, set `Status: resolved`, then append a context pointer (gist + link) to the map's Decisions-so-far in `map.md`. diff --git a/.github/skills/mattpocock-setup-matt-pocock-skills/triage-labels.md b/.github/skills/mattpocock-setup-matt-pocock-skills/triage-labels.md new file mode 100644 index 00000000..b716855d --- /dev/null +++ b/.github/skills/mattpocock-setup-matt-pocock-skills/triage-labels.md @@ -0,0 +1,15 @@ +# Triage Labels + +The skills speak in terms of five canonical triage roles. This file maps those roles to the actual label strings used in this repo's issue tracker. + +| Label in mattpocock/skills | Label in our tracker | Meaning | +| -------------------------- | -------------------- | ---------------------------------------- | +| `needs-triage` | `needs-triage` | Maintainer needs to evaluate this issue | +| `needs-info` | `needs-info` | Waiting on reporter for more information | +| `ready-for-agent` | `ready-for-agent` | Fully specified, ready for an AFK agent | +| `ready-for-human` | `ready-for-human` | Requires human implementation | +| `wontfix` | `wontfix` | Will not be actioned | + +When a skill mentions a role (e.g. "apply the AFK-ready triage label"), use the corresponding label string from this table. + +Edit the right-hand column to match whatever vocabulary you actually use. diff --git a/.github/skills/mattpocock-tdd/SKILL.md b/.github/skills/mattpocock-tdd/SKILL.md new file mode 100644 index 00000000..0bb28376 --- /dev/null +++ b/.github/skills/mattpocock-tdd/SKILL.md @@ -0,0 +1,36 @@ +--- +name: mattpocock-tdd +description: Test-driven development. Use when the user wants to build features or fix bugs test-first, mentions "red-green-refactor", or wants integration tests. +--- + +# Test-Driven Development + +TDD is the red → green loop. This skill is the reference that makes that loop produce tests worth keeping: what a good test is, where tests go, the anti-patterns, and the rules of the loop. Every section applies on every cycle — consult them before and during the loop, not after. + +When exploring the codebase, read `CONTEXT.md` (if it exists) so test names and interface vocabulary match the project's domain language, and respect ADRs in the area you're touching. + +## What a good test is + +Tests verify behavior through public interfaces, not implementation details. Code can change entirely; tests shouldn't. A good test reads like a specification — "user can checkout with valid cart" tells you exactly what capability exists — and survives refactors because it doesn't care about internal structure. + +See [tests.md](tests.md) for examples and [mocking.md](mocking.md) for mocking guidelines. + +## Seams — where tests go + +A **seam** is the public boundary you test at: the interface where you observe behavior without reaching inside. Tests live at seams, never against internals. + +**Test only at pre-agreed seams.** Before writing any test, write down the seams under test and confirm them with the user. No test is written at an unconfirmed seam. You can't test everything — agreeing the seams up front is how testing effort lands on the critical paths and complex logic instead of every edge case. + +Ask: "What's the public interface, and which seams should we test?" + +## Anti-patterns + +- **Implementation-coupled** — mocks internal collaborators, tests private methods, or verifies through a side channel (querying the database instead of using the interface). The tell: the test breaks when you refactor but behavior hasn't changed. +- **Tautological** — the assertion recomputes the expected value the way the code does (`expect(add(a, b)).toBe(a + b)`, a snapshot derived by hand the same way, a constant asserted equal to itself), so it passes by construction and can never disagree with the code. Expected values must come from an independent source of truth — a known-good literal, a worked example, the spec. +- **Horizontal slicing** — writing all tests first, then all implementation. Bulk tests verify _imagined_ behavior: you test the _shape_ of things rather than user-facing behavior, the tests go insensitive to real changes, and you commit to test structure before understanding the implementation. Work in **vertical slices** instead — one test → one implementation → repeat, each test a **tracer bullet** that responds to what the last cycle taught you. + +## Rules of the loop + +- **Red before green.** Write the failing test first, then only enough code to pass it. Don't anticipate future tests or add speculative features. +- **One slice at a time.** One seam, one test, one minimal implementation per cycle. +- **Refactoring is not part of the loop.** It belongs to the review stage (see the `mattpocock-code-review` skill), not the red → green implementation cycle. diff --git a/.github/skills/mattpocock-tdd/agents/openai.yaml b/.github/skills/mattpocock-tdd/agents/openai.yaml new file mode 100644 index 00000000..651b838a --- /dev/null +++ b/.github/skills/mattpocock-tdd/agents/openai.yaml @@ -0,0 +1,3 @@ +interface: + display_name: "TDD" + short_description: "Test-driven red-green-refactor" diff --git a/.github/skills/mattpocock-tdd/mocking.md b/.github/skills/mattpocock-tdd/mocking.md new file mode 100644 index 00000000..71cbfee6 --- /dev/null +++ b/.github/skills/mattpocock-tdd/mocking.md @@ -0,0 +1,59 @@ +# When to Mock + +Mock at **system boundaries** only: + +- External APIs (payment, email, etc.) +- Databases (sometimes - prefer test DB) +- Time/randomness +- File system (sometimes) + +Don't mock: + +- Your own classes/modules +- Internal collaborators +- Anything you control + +## Designing for Mockability + +At system boundaries, design interfaces that are easy to mock: + +**1. Use dependency injection** + +Pass external dependencies in rather than creating them internally: + +```typescript +// Easy to mock +function processPayment(order, paymentClient) { + return paymentClient.charge(order.total); +} + +// Hard to mock +function processPayment(order) { + const client = new StripeClient(process.env.STRIPE_KEY); + return client.charge(order.total); +} +``` + +**2. Prefer SDK-style interfaces over generic fetchers** + +Create specific functions for each external operation instead of one generic function with conditional logic: + +```typescript +// GOOD: Each function is independently mockable +const api = { + getUser: (id) => fetch(`/users/${id}`), + getOrders: (userId) => fetch(`/users/${userId}/orders`), + createOrder: (data) => fetch('/orders', { method: 'POST', body: data }), +}; + +// BAD: Mocking requires conditional logic inside the mock +const api = { + fetch: (endpoint, options) => fetch(endpoint, options), +}; +``` + +The SDK approach means: +- Each mock returns one specific shape +- No conditional logic in test setup +- Easier to see which endpoints a test exercises +- Type safety per endpoint diff --git a/.github/skills/mattpocock-tdd/tests.md b/.github/skills/mattpocock-tdd/tests.md new file mode 100644 index 00000000..7ab86479 --- /dev/null +++ b/.github/skills/mattpocock-tdd/tests.md @@ -0,0 +1,77 @@ +# Good and Bad Tests + +## Good Tests + +**Integration-style**: Test through real interfaces, not mocks of internal parts. + +```typescript +// GOOD: Tests observable behavior +test("user can checkout with valid cart", async () => { + const cart = createCart(); + cart.add(product); + const result = await checkout(cart, paymentMethod); + expect(result.status).toBe("confirmed"); +}); +``` + +Characteristics: + +- Tests behavior users/callers care about +- Uses public API only +- Survives internal refactors +- Describes WHAT, not HOW +- One logical assertion per test + +## Bad Tests + +**Implementation-detail tests**: Coupled to internal structure. + +```typescript +// BAD: Tests implementation details +test("checkout calls paymentService.process", async () => { + const mockPayment = jest.mock(paymentService); + await checkout(cart, payment); + expect(mockPayment.process).toHaveBeenCalledWith(cart.total); +}); +``` + +Red flags: + +- Mocking internal collaborators +- Testing private methods +- Asserting on call counts/order +- Test breaks when refactoring without behavior change +- Test name describes HOW not WHAT +- Verifying through external means instead of interface + +```typescript +// BAD: Bypasses interface to verify +test("createUser saves to database", async () => { + await createUser({ name: "Alice" }); + const row = await db.query("SELECT * FROM users WHERE name = ?", ["Alice"]); + expect(row).toBeDefined(); +}); + +// GOOD: Verifies through interface +test("createUser makes user retrievable", async () => { + const user = await createUser({ name: "Alice" }); + const retrieved = await getUser(user.id); + expect(retrieved.name).toBe("Alice"); +}); +``` + +**Tautological tests**: Expected value restates the implementation, so the test passes by construction. + +```typescript +// BAD: Expected value is recomputed the way the code computes it +test("calculateTotal sums line items", () => { + const items = [{ price: 10 }, { price: 5 }]; + const expected = items.reduce((sum, i) => sum + i.price, 0); + expect(calculateTotal(items)).toBe(expected); +}); + +// GOOD: Expected value is an independent, known literal +test("calculateTotal sums line items", () => { + expect(calculateTotal([{ price: 10 }, { price: 5 }])).toBe(15); +}); +``` diff --git a/.github/skills/mattpocock-to-spec/SKILL.md b/.github/skills/mattpocock-to-spec/SKILL.md new file mode 100644 index 00000000..7431b9de --- /dev/null +++ b/.github/skills/mattpocock-to-spec/SKILL.md @@ -0,0 +1,75 @@ +--- +name: mattpocock-to-spec +description: Turn the current conversation into a spec and publish it to the project issue tracker — no interview, just synthesis of what you've already discussed. +disable-model-invocation: true +--- + +This skill takes the current conversation context and codebase understanding and produces a spec (you may know this document as a PRD). Do NOT interview the user — just synthesize what you already know. + +The issue tracker and triage label vocabulary should have been provided to you — run `/mattpocock-setup-matt-pocock-skills` if not. + +## Process + +1. Explore the repo to understand the current state of the codebase, if you haven't already. Use the project's domain glossary vocabulary throughout the spec, and respect any ADRs in the area you're touching. + +2. Sketch out the seams at which you're going to test the feature. Existing seams should be preferred to new ones. Use the highest seam possible. If new seams are needed, propose them at the highest point you can. The fewer seams across the codebase, the better - the ideal number is one. + +Check with the user that these seams match their expectations. + +3. Write the spec using the template below, then publish it to the project issue tracker. Apply the `ready-for-agent` triage label - no need for additional triage. + + + +## Problem Statement + +The problem that the user is facing, from the user's perspective. + +## Solution + +The solution to the problem, from the user's perspective. + +## User Stories + +A LONG, numbered list of user stories. Each user story should be in the format of: + +1. As an , I want a , so that + + +1. As a mobile bank customer, I want to see balance on my accounts, so that I can make better informed decisions about my spending + + +This list of user stories should be extremely extensive and cover all aspects of the feature. + +## Implementation Decisions + +A list of implementation decisions that were made. This can include: + +- The modules that will be built/modified +- The interfaces of those modules that will be modified +- Technical clarifications from the developer +- Architectural decisions +- Schema changes +- API contracts +- Specific interactions + +Do NOT include specific file paths or code snippets. They may end up being outdated very quickly. + +Exception: if a prototype produced a snippet that encodes a decision more precisely than prose can (state machine, reducer, schema, type shape), inline it within the relevant decision and note briefly that it came from a prototype. Trim to the decision-rich parts — not a working demo, just the important bits. + +## Testing Decisions + +A list of testing decisions that were made. Include: + +- A description of what makes a good test (only test external behavior, not implementation details) +- Which modules will be tested +- Prior art for the tests (i.e. similar types of tests in the codebase) + +## Out of Scope + +A description of the things that are out of scope for this spec. + +## Further Notes + +Any further notes about the feature. + + diff --git a/.github/skills/mattpocock-to-spec/agents/openai.yaml b/.github/skills/mattpocock-to-spec/agents/openai.yaml new file mode 100644 index 00000000..549e6f76 --- /dev/null +++ b/.github/skills/mattpocock-to-spec/agents/openai.yaml @@ -0,0 +1,5 @@ +interface: + display_name: "To Spec" + short_description: "Turn a conversation into a spec" +policy: + allow_implicit_invocation: false diff --git a/.github/skills/mattpocock-wayfinder/SKILL.md b/.github/skills/mattpocock-wayfinder/SKILL.md new file mode 100644 index 00000000..456cc0b7 --- /dev/null +++ b/.github/skills/mattpocock-wayfinder/SKILL.md @@ -0,0 +1,128 @@ +--- +name: mattpocock-wayfinder +description: Plan a huge chunk of work — more than one agent session can hold — as a shared map of decision tickets on your issue tracker, and resolve them one at a time until the way to the destination is clear. +disable-model-invocation: true +--- + +A loose idea has arrived — too big for one agent session, and wrapped in fog: the way from here to the **destination** isn't visible yet. Wayfinding is about finding that way, not charging at the destination. This skill charts the way as a **shared map** on the repo's issue tracker, then works its **decision tickets** — questions whose resolution is a decision, not slices of a build to execute — one at a time until the route is clear. + +The destination varies per effort, and naming it is the first act of charting — it shapes every ticket. It might be a spec to hand off and iterate on, a decision to lock before planning starts, or a change made in place like a data-structure migration. The map is domain-agnostic — engineering work, course content, whatever fits the shape. + +## Plan, don't do + +Wayfinder is **planning** by default: each ticket resolves a decision, and the map is done when the way is clear — nothing left to decide before someone goes and does the thing. The pull to just do the work is usually the signal you've reached the edge of the map and it's time to hand off. An effort can override this in its **Notes** — carrying execution into the map itself — but absent that, produce decisions, not deliverables. + +## Refer by name + +Every map and ticket is an issue, so it has a **name** — its title. In everything the human reads — narration, the map's Decisions-so-far — refer to it by that name, never by a bare id, number, or slug. A wall of `#42, #43, #44` is illegible; names read at a glance. The id and URL don't vanish — a name wraps its link — but they ride *inside* the name, never stand in for it. + +## The Map + +The map is a single issue on this repo's issue tracker, labelled `wayfinder:map` — the canonical artifact. Its tickets are child issues of the map. + +The map is an **index**, not a store. It lists the decisions made and points at the tickets that hold their detail; a decision lives in exactly one place — its ticket — so the map never restates it, only gists it and links. + +**Where the map, its child tickets, blocking, and frontier queries physically live is tracker-specific.** The issue tracker should have been provided to you — run `/mattpocock-setup-matt-pocock-skills` if not. Consult the tracker doc's "Wayfinding operations" section for how _this_ repo expresses them. If no tracker has been provided, default to the local-markdown tracker. + +### The map body + +The whole map at low resolution, loaded once per session. Open tickets are **not** listed — they are open child issues, found by query. + +```markdown +## Destination + + + +## Notes + + + +## Decisions so far + + + +- [](link) — + +## Not yet specified + + + +## Out of scope + + +``` + +### Tickets + +Each ticket is a **child issue** of the map; the tracker's issue id is its identity. Its body is the question, sized to one 100K token agent session: + +```markdown +## Question + + +``` + +Each ticket carries a `wayfinder:` label — one of `mattpocock-research`, `prototype`, `grilling`, `task` (see [Ticket Types](#ticket-types)). + +A session **claims** a ticket by assigning it to the dev driving the map, **first**, before any work, so concurrent sessions skip it. That assignee _is_ the claim: an open, unassigned ticket is unclaimed. + +Blocking uses the tracker's **native** dependency relationship — essential because it renders the frontier _visually_ in the tracker's own UI, so the human sees what's takeable without opening the map. Only a tracker that lacks native blocking falls back to a body convention. A ticket is **unblocked** when every ticket blocking it is closed; the **frontier** is the open, unblocked, unclaimed children — the edge of the known. + +The answer isn't part of the body — it's recorded on resolution (see [Work through the map](#work-through-the-map)). Assets created while resolving a ticket are linked from the issue, not pasted in. + +## Ticket Types + +Every ticket is either **HITL** — human in the loop, worked *with* a human who speaks for themselves — or **AFK**, driven by the agent alone. A HITL ticket only resolves through that live exchange; the agent never stands in for the human's side of it (a grilling agent that answers its own questions has broken this). + +- **Research** (AFK): Reading documentation, third-party APIs, or local resources like knowledge bases to surface a fact a decision waits on. Resolved by a `/mattpocock-research` **subagent**. Use when knowledge outside the current working directory is required. +- **Prototype** (HITL): Raise the fidelity of the discussion by making a cheap, rough, concrete artifact to react to — an outline, a rough take, a stub, or UI/logic code via the /prototype skill. Links the prototype as an asset. Use when "how should it look" or "how should it behave" is the key question. +- **Grilling** (HITL): Conversation via the /grill-me and /mattpocock-domain-modeling skills, one question at a time. The default case. +- **Task** (HITL or AFK): Manual work that must happen before a *decision* can be made — nothing to decide, prototype, or research, but the discussion is blocked until it's done. Signing up for a service so its API can be judged, provisioning access, moving data so its shape can be seen. This is the one type that *does* rather than decides — and it earns its place by unblocking a decision, not by delivering the destination. The agent drives it alone where it can (AFK); otherwise it hands the human a precise checklist (HITL). Resolved when the work is done; the answer records what was done and any resulting facts (credentials location, new URLs, row counts) later tickets depend on. + +## Fog of war + +The map is _deliberately_ incomplete: don't chart what you can't yet see. Beyond the live tickets lies the **fog of war** — the dim view of decisions and investigations you can tell are coming but can't yet pin down, because they hang on questions still open. Resolving a ticket clears the fog ahead of it, graduating whatever's now specifiable into fresh tickets — one at a time, until the way to the destination is clear and no tickets remain. + +The map's **Not yet specified** section is where that dim view is written down: the suspected question, the area to revisit later. It's the undiscovered frontier _toward_ the destination — everything here is in scope, just not sharp enough to ticket. Write as loosely or as fully as the view allows; it doubles as a signpost for collaborators reading where the effort is headed. + +**Fog or ticket?** The test is whether you can state the question precisely now — _not_ whether you can answer it now. + +- **Ticket when** the question is already sharp — even if it's blocked and you can't act on it yet. +- **Not yet specified when** you can't yet phrase it that sharply. Don't pre-slice the fog into ticket-sized pieces: it's coarser than a ticket, and one patch may graduate into several tickets, or none, once the frontier reaches it. + +**Not yet specified** excludes what's already decided (Decisions so far), what's already a live ticket, and what's out of scope (the next section). + +## Out of scope + +Fog only ever gathers _toward_ the destination. The destination fixes the scope, so work beyond it is **out of scope** — it isn't fog, and it doesn't belong in **Not yet specified**. It gets its own **Out of scope** section on the map: work you've consciously ruled out of _this_ effort. Scope, not sharpness, lands it here. + +Out-of-scope work never graduates — the frontier stops at the destination — so it returns only if the destination is redrawn, and then as a fresh effort, not a resumption. + +Ruling something out of scope is a scoping act, not a step on the route. When a ticket that already exists turns out to sit past the destination — mis-scoped in while charting, or exposed by a resolution — **close it** (a closed ticket is unambiguously off the frontier) and leave one line in the **Out of scope** section: the gist plus why it's out of scope, linking the closed ticket. It stays out of **Decisions so far**, which records the route actually walked — a scope boundary isn't a step on it. + +## Invocation + +Two modes. Either way, **never resolve more than one ticket per session** — with the exception of research tickets. + +### Chart the map + +User invokes with a loose idea. + +1. **Name the destination.** Run a `/grill-me` and `/mattpocock-domain-modeling` session to pin down what this map is finding its way to — the spec, decision, or change. The destination fixes the scope, so it's settled first. +2. **Map the frontier.** Grill again, **breadth-first** this time: fan out across the whole space rather than deep on any one thread, surfacing the open decisions and the first steps takeable now. **If this surfaces no fog** — the way to the destination is already clear, the whole journey small enough for one session — you don't need a map. Stop and ask the user how they'd like to proceed. +3. **Create the map** (label `wayfinder:map`): Destination and Notes filled in, Decisions-so-far empty, the fog sketched into **Not yet specified**. +4. **Create the tickets you can specify now** as child issues of the map — then wire blocking edges in a **second pass** (issues need ids before they can reference each other). Wiring sorts them into the frontier and the blocked; everything you can't yet specify stays in the fog — the **Not yet specified** section. +5. **Fire the research subagents.** For each `mattpocock-research` ticket you just created, spin up a `/mattpocock-research` subagent to resolve it in parallel, capturing its findings on a throwaway `research/` branch with a context pointer from the ticket. +6. Stop — charting is one session's work; it hand-resolves nothing. + +### Work through the map + +User invokes with a map (URL or number). A ticket is **optional** — without one, you pick the next decision, not the user. + +1. Load the **map** — the low-res view, not every ticket body. +2. Choose the ticket. If the user named one, use it. Otherwise take the first frontier ticket in order. **Claim it**: assign it to yourself before any work. +3. Resolve it — **zoom as needed**: fetch the full body of any related or closed ticket on demand; invoke the skills the `## Notes` block names. If in doubt, use `/grill-me` and `/mattpocock-domain-modeling`. +4. Record the resolution: post the answer as a **resolution comment**, **close** the issue, and **append a context pointer** to the map's Decisions-so-far. +5. Add newly-surfaced tickets (create-then-wire); graduate any fog the answer has made specifiable, clearing each graduated patch from **Not yet specified** so it lives only as its new ticket. If the answer reveals a ticket — this one or another — sits beyond the destination, **rule it out of scope** rather than resolving it on the route. If the decision invalidates other parts of the map, update or delete those tickets. + +The user may run unblocked tickets in parallel, so expect other sessions to be editing the tracker concurrently. diff --git a/.github/skills/mattpocock-wayfinder/agents/openai.yaml b/.github/skills/mattpocock-wayfinder/agents/openai.yaml new file mode 100644 index 00000000..b3754475 --- /dev/null +++ b/.github/skills/mattpocock-wayfinder/agents/openai.yaml @@ -0,0 +1,5 @@ +interface: + display_name: "Wayfinder" + short_description: "Map a large effort as decision tickets" +policy: + allow_implicit_invocation: false diff --git a/.github/skills/mattpocock-writing-great-skills/GLOSSARY.md b/.github/skills/mattpocock-writing-great-skills/GLOSSARY.md new file mode 100644 index 00000000..935459f8 --- /dev/null +++ b/.github/skills/mattpocock-writing-great-skills/GLOSSARY.md @@ -0,0 +1,201 @@ +# Glossary — Building Great Skills + +The domain model for what makes a skill great. A skill exists to wrangle determinism out of a stochastic system; the root virtue is **Predictability**, and every term below is a lever on it. This is the disclosed reference for [`mattpocock-writing-great-skills`](SKILL.md). + +The terms are grouped by axis: **Invocation** (how a skill is reached), **Information Hierarchy** (how its content is arranged), **Steering** (how the agent's runtime behaviour is shaped), and **Pruning** (how it is kept lean). Each **failure mode** lives beside the lever that cures it, tagged _failure mode_. + +**Bold terms** in any definition are themselves defined in this glossary; find them by their heading. + +## Predictability + +The degree to which a skill makes the agent behave the same _way_ on every run — the same process, not the same output (a brainstorming skill should _predictably_ diverge; its tokens vary, its behaviour doesn't). The root virtue every other term serves — cost and maintainability are symptoms of it, not rivals. + +_Avoid_: consistency, reliability, robustness, output-determinism + +## Invocation + +How a skill is reached — and the two loads you pay for the choice. + +### Model-Invoked + +A skill that keeps its **description** field, so the agent can see it and fire it autonomously — and the human can still type its name, so model-invocation always _includes_ user reach. There is no model-only state: a description only ever _adds_ agent discovery, never removes the human's. Pays a permanent **context load** on every turn in exchange for that discoverability. Reachable by other skills, because the description that makes it agent-discoverable makes it invocable. A model-invoked skill whose content is all **reference** is also one home for shared reference: another skill can invoke it, so reference needed by several skills lives in one place. Pick model-invocation only when the agent must reach the skill on its own; if it never fires except by hand, drop the description and pay no context load. + +_Avoid_: ability, tool, capability + +### User-Invoked + +A skill with its **description** stripped — invisible to the agent and reachable only by the human typing its name (user-_only_, where **model-invoked** is user-_and-agent_). Trades agent-discoverability for zero **context load**. Because it has no description, nothing but the human can reach it: no other skill can fire it. + +_Avoid_: procedure, workflow, command + +### Description + +The skill's machine-readable trigger, and the one **context pointer** a **model-invoked** skill is forced to keep loaded at all times. Its mere presence _is_ the invocation axis: keep it and the skill is model-invoked (and reachable by other skills); delete it and the skill is **user-invoked**, reachable only by the human. The source of a model-invoked skill's **context load**. + +_Avoid_: frontmatter, summary + +### Context Pointer + +A reference held in the agent's context that names some out-of-context material and encodes the condition for reaching it. The **description** is the top-level context pointer (context window → skill); pointers to disclosed files are the same object one level down. Its wording, not the target, decides _when_ the agent reaches — and _how reliably_. A must-have target behind a weakly worded pointer is a variance bug: fix the wording first, and inline the material only if sharpening fails. + +_Avoid_: link, reference, import + +### Context Load + +The cost a **model-invoked** skill imposes on the agent's context window — its **description**, always loaded, spending both tokens and attention. What **user-invoked** skills escape by having no description, and the brake on splitting into more model-invoked skills. + +_Avoid_: token cost, context bloat + +### Cognitive Load + +The cost a **user-invoked** skill imposes on the human — what they must hold in their head: which skills exist and when to reach for each (the human is the index). What **model-invocation** removes by being agent-discoverable, and the brake on splitting into more user-invoked skills. Not a cost to minimise: it is the price of human agency, the reason some skills stay user-invoked. Spend it where human judgement matters; remove it where it does not. + +_Avoid_: human index, burden, overhead + +### Router Skill + +A **user-invoked** skill whose job is to point at your other user-invoked skills — naming each and when to reach for it — so the human has one skill to remember instead of many. It can only hint, never fire them: user-invoked skills have no **description**, so nothing but the human can reach them. The cure for **cognitive load** when user-invoked skills multiply. + +_Avoid_: dispatcher, menu, registry, index, router procedure + +### Granularity + +How finely you divide skills. Finer division spends one of the two loads: more **model-invoked** skills spend **context load** (more descriptions crowding the window and competing for attention); more **user-invoked** skills spend **cognitive load** (more for the human to remember and reach for). Two cuts guide the division. By **invocation**, split off a model-invoked skill where you have a distinct **leading word** to trigger it — a trigger word you actually use in your prompts. By **sequence**, split a run of **steps** where a step's **post-completion steps** need hiding, since isolating it in its own context clears what follows. Beware the reverse: merging sequences exposes each step's post-completion steps to what follows, inviting premature completion. + +_Avoid_: chunking, modularity + +## Information Hierarchy + +How a skill's content is arranged, and how far down the ladder each piece sits. + +### Information Hierarchy + +A skill's content ranked by how immediately the agent needs it — a single ladder, produced by two cuts: in-file or behind a pointer, and step or reference. The rungs: + +- **Steps** — in-file, primary +- **Reference**, in-file — secondary +- **Reference**, disclosed — behind a **context pointer** + +A skill with no **steps** uses just the bottom two rungs — often a legitimately flat peer-set (e.g. every rule of a review on one rung), which is a fine arrangement, not a smell. The hierarchy is independent of invocation: a skill can be model- or user-invoked whether it is all steps, all reference, or both. When a skill has steps, in-file reference that should be disclosed buries them and turns attending to them into a coin-flip — a variance lever, not just a legibility one. Keep the top of the ladder legible; push down it whatever you can. + +_Avoid_: structure, organization, layout + +### Steps + +The ordered actions the agent performs — when a skill has them, the primary tier of its content, and the part that earns its place in SKILL.md. Not every skill has steps: a skill can be all steps (`mattpocock-tdd`), all **reference** (a review), or both, independent of invocation. Every step ends on a **completion criterion**, clear or vague. + +_Avoid_: workflow, instructions, choreography + +### Reference + +Material the agent refers to on demand — definitions, facts, parameters, examples, conditional instructions. When a skill has **steps** it is secondary to them; when a skill has none it is the entire content; or it lives outside any skill entirely — see **External Reference**. Reached via **context pointers**, and the prime candidate for **progressive disclosure**. + +_Avoid_: supporting material, docs, background + +### External Reference + +**Reference** that lives outside the skill system — a plain file, no **description**, no **steps**, not invocable — that any skill can point at. The home for shared reference that needn't fire on its own, and the only shared home two **user-invoked** skills can use, since neither has a description and so neither can fire the other. + +_Avoid_: doc, resource, knowledge base + +### Progressive Disclosure + +Moving **reference** down the ladder — out of SKILL.md and behind a **context pointer** — so the top stays legible. Not primarily a token optimisation; it is how the **information hierarchy** is protected. Licensed by **branching**: disclose what only some branches need, inline what every path needs, and if a pointer fires unreliably on must-have material, sharpen its wording, and pull it back inline only if that fails. + +_Avoid_: lazy loading, chunking + +### Co-location + +Keeping the material an agent needs at once in one place — a concept's definition, rules, and caveats under a single heading, not scattered across the file — so reading one part brings its neighbours with it. The within-file companion to the **Information Hierarchy**: the hierarchy ranks _how far down_ a piece sits; co-location decides _what sits beside it_ once there. There is no formula for the right format of a body of **reference**; the test is that a skill should read like documentation written for the agent, and grouped material reads that way where scattered material does not. Distinct from **Duplication**: that repeats one meaning in two places, where scattering fragments a single meaning across many. + +_Avoid_: grouping, clustering, cohesion + +### Sprawl + +_Failure mode._ A skill that is simply too long — too many lines in SKILL.md — independent of whether they are stale or repeated. Even an all-live, all-unique skill can sprawl. It costs readability (the agent wades through more before it can act, and attention thins across the excess), maintainability (every extra line is one more to keep **relevant**), and tokens. The cure is the **information hierarchy**: push **reference** down behind **context pointers**, and split by **branch** or sequence so each path carries only what it needs. Distinct from **sediment** (length from stale accumulation) and **duplication** (length from repeated meaning) — sprawl is length itself, whatever its cause. + +_Avoid_: bloat, length, size, verbosity + +## Steering + +The levers that shape the agent's runtime behaviour toward **Predictability**. + +### Branch + +A distinct way a skill can be invoked — a case the skill handles — so different runs take different paths through it. A skill with many steps may carry many branches; a linear one has none. + +_Avoid_: path, case, fork + +### Leading Word + +A compact concept — also called a _Leitwort_ — already living in the model's pretraining, that the agent thinks with while running the skill. It encodes a behavioural principle in the fewest possible tokens by invoking priors the model already holds (e.g. _lesson_, _proximal zone of development_, _fog of war_, _tracer bullets_). Repeated as a token, never as a sentence, it accumulates a distributed definition across the skill and anchors a whole region of behaviour. Coining your own works if you define it clearly, but a made-up word recruits no priors — you pay in definition tokens what a pretrained word gives free. Reach for an existing word first. + +A leading word serves **predictability** twice. In the body it anchors **execution** — the agent reaches for the same behaviour every time the concept appears, and inside flat reference it focuses attention on a class of thing to look for, recruiting the right checks each run. In the **description** it anchors **invocation** — and not only within the skill: when the same word lives in your prompts, your docs, and your codebase, the agent links that shared language to the skill and fires it more reliably. Word a description with the leading words you actually use when you want the skill. + +_Avoid_: keyword, term, motif + +### Completion Criterion + +The condition that tells the agent a unit of work is done — the target it judges against. Two properties make it a lever, not just a quality. Its **clarity** (can the agent tell done from not-done?) resists **premature completion** — a vague bound ("understanding reached") lets the agent declare done and slip to the next step; this axis needs _steps_ to bite, since premature completion is a between-steps failure. Its **demand** (how much it requires) sets **legwork** — "every modified model accounted for" forces thorough work where "produce a change list" does not — and this axis is _not_ step-bound: it can bind a body of flat reference too, which is how a skill with no steps still carries an exhaustiveness bar ("every rule applied"). The strongest criteria are both checkable and exhaustive. + +_Avoid_: done condition, exit condition, stopping rule + +### Legwork + +The work an agent does behind the scenes within a single step — reading files, exploring the codebase, making changes, digging up what it needs rather than offloading to the user. It lives below the step structure: never written as its own step, latent in the wording, controlled by the agent rather than the skill. The within-step counterpart to **post-completion steps**' across-step pull. Raised by a **leading word** (_comprehensive_, _thorough_) or a **completion criterion** that demands the work be exhaustive — including the demand axis applied to flat reference, which is what drives a skill of flat reference to cover all its rungs. Goes thin either when that demand is missing or when **premature completion** cuts the step short. + +_Avoid_: scope, effort, diligence, coverage + +### Post-Completion Steps + +The **steps** that follow the current step. Visible, they pull the agent forward into **premature completion** — the more it sees, the stronger the tug; the defence is to hide them by splitting the sequence of steps into two. + +_Avoid_: horizon, fog of war, lookahead + +### Premature Completion + +_Failure mode._ Ending the current step before it is genuinely done, because the agent's attention slips to being done rather than to the work. A between-steps failure: it needs **steps** to occur — a skill with no steps that quits early isn't premature completion but thin **legwork** under an unmet demand. A tug-of-war between two forces: visible **post-completion steps** (the pull forward) and the **completion criterion**'s clarity (the resistance — a sharp, checkable bar holds; a vague one gives way). Fuzziness is the necessary condition: a sharp bound resists the pull no matter how many later steps are visible, so a step that never rushes needs no defending. Two levers hold a step that does, but reach for them in order: **sharpen the bound first** — it is local and cheap. Only when the criterion is irreducibly fuzzy _and_ you actually observe the rush do you **hide the later steps** — and hiding only works across a real context boundary (a user-invoked hand-off or a subagent dispatch; an inline model-invoked call leaves the later steps in context and clears nothing). One cause of thin legwork, but distinct from it: legwork can be thin even when a step runs to full completion. + +_Avoid_: premature closure, the rush, rushing, shortcutting + +### Negation + +_Failure mode._ Steering by prohibition — telling the agent what _not_ to do — which drags the forbidden behaviour into context and makes it _more_ available, not less. _Don't think of an elephant_, and the elephant is all there is; _never write verbose comments_, and verbosity is the pattern the agent has just read. The negation is a weak modifier the strongly-activated concept overruns, so the ban half-reads as an instruction to do the thing. Its **leading word** is the _elephant_: whatever a prohibition names into the frame. Cure: prompt the **positive** — describe the target behaviour ("write one-line comments") so the banned one is never spoken. A prohibition earns its place only as a hard guardrail on a behaviour you cannot phrase positively; even then, pair it with the positive target so attention lands on what to do. + +_Avoid_: ironic rebound, don't-prompting, the pink elephant + +## Pruning + +Keeping a skill lean — each remedy paired with the failure it cures. + +### Single Source of Truth + +The desired state where each meaning lives in exactly one authoritative place, so a change to the skill's behaviour is a change in one place. **Duplication** is its violation. + +_Avoid_: home, canonical location + +### Duplication + +_Failure mode._ The same meaning given more than one **single source of truth**. It costs maintenance (change one place, you must change the others), costs tokens, and inflates prominence — repeating a meaning weights it on the ladder past its real rank. The accidental inverse of a **leading word**, which raises attention on purpose by repeating a token, never the meaning. + +_Avoid_: repetition, redundancy + +### Relevance + +Whether a line still bears on what the skill does — the lens for what to keep. A line loses relevance either by never bearing on the task (mere exposition, or a **branch** that should be disclosed) or by going stale: drifting out of date as the behaviour or world it describes changes. Shorter skills are easier to keep relevant, because each line is cheaper to check. Distinct from **no-op**: relevance asks whether a line bears on the task, not whether it changes behaviour. + +_Avoid_: load-bearing, staleness, freshness + +### Sediment + +_Failure mode._ Layers of old content that settle in a skill and are never cleared, because adding feels safe and removing feels risky — so stale and irrelevant lines accumulate and you must core down through them to find what is still live. The default fate of any skill without a pruning discipline; the slow erosion of **relevance**, as opposed to **duplication**'s repeated meaning. + +_Avoid_: accretion, bloat, cruft, rot + +### No-Op + +_Failure mode._ An instruction that changes nothing because the model already does it by default — you pay load to tell the agent what it would do anyway. The test: does a line change behaviour versus the default? A line can be perfectly **relevant** and still be a no-op. The same priors that make a **leading word** free make a no-op worthless. + +A leading word is a _technique_; No-Op is a _verdict_ on a line — and they cross. A leading word too weak to beat the default is a no-op (_be thorough_ when the agent is already thorough-ish), and the fix is a stronger word that passes the verdict (_relentless_), not a different technique. So the No-Op test — does it change behaviour versus the default? — is also how you grade whether a leading word is earning its repetitions. This is model-relative, not reader-relative: two people disagreeing over whether a line is a no-op disagree about the default, and settle it by running the skill, not by debate. + +_Avoid_: redundant instruction, restating the obvious, belaboring diff --git a/.github/skills/mattpocock-writing-great-skills/SKILL.md b/.github/skills/mattpocock-writing-great-skills/SKILL.md new file mode 100644 index 00000000..a1687969 --- /dev/null +++ b/.github/skills/mattpocock-writing-great-skills/SKILL.md @@ -0,0 +1,82 @@ +--- +name: mattpocock-writing-great-skills +description: Predictability review and revision stage applied after authoring. Loaded by internal-skill-creator after the Anthropic authoring stage when delegating skill work. +--- + +A skill exists to wrangle determinism out of a stochastic system. **Predictability** — the agent taking the same _process_ every run, not producing the same output — is the root virtue; every lever below serves it. + +**Bold terms** are defined in [`GLOSSARY.md`](GLOSSARY.md); look them up there for the full meaning. + +## Invocation + +Two choices, trading different costs: + +- A **model-invoked** skill keeps a **description**, so the agent can fire it autonomously _and_ other skills can reach it (you can still type its name too). It contributes to **context load** — the description sits in the window every turn. Mechanics: omit `disable-model-invocation`, and write a model-facing description with rich trigger phrasing ("Use when the user wants…, mentions…"). +- A **user-invoked** skill strips the description from the agent's reach: only you, typing its name, can invoke it — and no other skill can. Zero context load, but it spends **cognitive load**: _you_ are the index that must remember it exists. Mechanics: set `disable-model-invocation: true`; the `description` becomes human-facing — a one-line summary, trigger lists stripped. + +Pick model-invocation only when the agent must reach the skill on its own, or another skill must. If it only ever fires by hand, make it user-invoked and pay no context load. + +When user-invoked skills multiply past what you can remember, that piled-up cognitive load is cured by a **router skill**: one user-invoked skill that names the others and when to reach for each. + +## Writing the description + +A model-invoked **description** does two jobs — state what the skill is, and list the **branches** that should trigger it. Every word increases **context load**, so a description earns even harder pruning than the body: + +- **Front-load the skill's leading word** — the description is where it does its invocation work. +- **One trigger per branch.** Synonyms that rename a single branch are **duplication** — "build features using TDD … asks for test-first development" is one branch written twice. Collapse them; keep only genuinely distinct branches. +- **Cut identity that's already in the body.** Keep the description to triggers, plus any "when another skill needs…" reach clause. + +## Information hierarchy + +A skill is built from two content types — **steps** and **reference** — that mix freely: a skill can be all steps, all reference, or both. The core decision is which to use and where each sits on the **information hierarchy**, a ladder ranked by how immediately the agent needs the material: + +1. **In-skill step** — an ordered action in `SKILL.md`, the primary tier: what the agent does, in order. Each step ends on a **completion criterion**, the condition that tells the agent the work is done. Make it _checkable_ (can the agent tell done from not-done?) and, where it matters, _exhaustive_ ("every modified model accounted for", not "produce a change list") — a vague criterion invites **premature completion**. +2. **In-skill reference** — a definition, rule, or fact in `SKILL.md`, consulted on demand. Often a legitimately flat peer-set (every rule of a review on one rung) — a fine arrangement, not a smell. _This skill is all reference._ +3. **External reference** — reference pushed out of `SKILL.md` into a separate file, reached by a **context pointer**, loaded only when the pointer fires. (Spans _disclosed_ reference — a sibling file like `GLOSSARY.md`, still part of the skill — through fully **external reference** that lives outside the skill system and any skill can point at.) + +A demanding completion criterion drives thorough **legwork** — the digging the agent does within the work — whether the skill has steps or not, since "every rule applied" binds flat reference just as "every step done" binds a sequence. + +Push too little down and the top bloats; push too much and you hide material the agent actually needs. That tension is the whole decision. + +**Progressive disclosure** is the move down the ladder — out of `SKILL.md` into a linked file — so the top stays legible. Mechanics: a linked `.md` file in the skill folder, named for what it holds (this skill discloses its full definitions to `GLOSSARY.md`). Some skills are used in more than one way, and each distinct way is a **branch** — different runs taking different paths through the skill. Branching is the cleanest disclosure test: inline what every branch needs, and push behind a pointer what only some branches reach. A **context pointer**'s _wording_, not its target, decides when and how reliably the agent reaches the material. + +Where the ladder decides _how far down_ a piece sits, **co-location** decides _what sits beside it_ once there: keep a concept's definition, rules, and caveats under one heading rather than scattered, so reading one part brings its neighbours with it. + +## When to split + +**Granularity** is how finely you divide skills, and each cut spends one of the two loads, so split only when the cut earns it. Two cuts: + +- **By invocation** — split off a **model-invoked** skill when you have a distinct **leading word** that should trigger it on its own, or another skill must reach it. You pay **context load** for the new always-loaded **description**, so that independent reach has to be worth it. +- **By sequence** — split a run of **steps** when the steps still ahead (a step's **post-completion steps**) tempt the agent to rush the one in front of it (**premature completion**). Keeping them out of view encourages the agent to do more **legwork** on the current task. + +## Pruning + +Keep each meaning in a **single source of truth**: one authoritative place, so changing the behaviour is a one-place edit. + +Check every line for **relevance**: does it still bear on what the skill does? + +Then hunt **no-ops** sentence by sentence, not just line by line: run the no-op test on each sentence in isolation, and when one fails, delete the whole sentence rather than trim words from it. Be aggressive — most prose that fails should go, not be rewritten. + +## Leading words + +A **leading word** is a compact concept already living in the model's pretraining that the agent thinks with while running the skill (e.g. _lesson_, _fog of war_, _tracer bullets_). Repeated throughout the text (though not necessarily - a strong leading word might only be needed once), it accumulates a distributed definition and anchors a whole region of behaviour in the fewest tokens, by recruiting priors the model already holds. + +It serves predictability twice. In the body it anchors _execution_: the agent reaches for the same behaviour every time the word appears. In the description it anchors _invocation_: when the same word lives in your prompts, docs, and code, the agent links that shared language to the skill and fires it more reliably. + +Hunt for opportunities to refactor skills to use leading words. A triad spelled out at three sites (**duplication**), a description spending a sentence to gesture at one idea — each is a passage begging to **collapse** into a single token. Examples include: + +- "fast, deterministic, low-overhead" -> _tight_ — one quality restated across a phase — into a single pretrained word (a _tight_ loop). +- "a loop you believe in" -> _red_ — converts a fuzzy gate into a binary observable state (the loop goes _red_ on the bug, or it doesn't). + +You win twice over: fewer tokens, _and_ a sharper hook for the agent to hang its thinking on. Assume every skill is carrying restatements that leading words retire — go find them. + +## Failure modes + +Use these to diagnose issues the user may be having with the skill. + +- **Premature completion** — ending a step before it's genuinely done, attention slipping to _being done_. Defence, in order: sharpen the completion criterion first (cheap, local); only if it is irreducibly fuzzy _and_ you observe the rush, hide the post-completion steps by splitting (the sequence cut). +- **Duplication** — the same meaning in more than one place. Costs maintenance and tokens, and inflates a meaning's prominence on the ladder past its real rank. +- **Sediment** — stale layers that settle because adding feels safe and removing feels risky. The default fate of any skill without a pruning discipline. +- **Sprawl** — a skill simply too long, even when every line is live and unique. Hurts readability and maintainability and wastes tokens. The cure is the ladder: disclose **reference** behind pointers, and split by **branch** or sequence so each path carries only what it needs. +- **No-op** — a line the model already obeys by default, so you pay load to say nothing. The test: does it change behaviour versus the default? A weak leading word (_be thorough_ when the agent is already thorough-ish) is a no-op; the fix is a stronger word (_relentless_), not a different technique. +- **Negation** — steering by prohibition backfires: _don't think of an elephant_ names the elephant and makes it more available, not less. Prompt the **positive** — state the target behaviour so the banned one is never spoken; keep a prohibition only as a hard guardrail you can't phrase positively, and even then pair it with what to do instead. diff --git a/.github/skills/mattpocock-writing-great-skills/agents/openai.yaml b/.github/skills/mattpocock-writing-great-skills/agents/openai.yaml new file mode 100644 index 00000000..d677a13f --- /dev/null +++ b/.github/skills/mattpocock-writing-great-skills/agents/openai.yaml @@ -0,0 +1,5 @@ +interface: + display_name: "Writing Great Skills" + short_description: "Principles for predictable skills" +policy: + allow_implicit_invocation: false diff --git a/.github/skills/openai-docx/LICENSE.txt b/.github/skills/openai-docs/LICENSE.txt similarity index 100% rename from .github/skills/openai-docx/LICENSE.txt rename to .github/skills/openai-docs/LICENSE.txt diff --git a/.github/skills/openai-docs/SKILL.md b/.github/skills/openai-docs/SKILL.md new file mode 100644 index 00000000..92bf2252 --- /dev/null +++ b/.github/skills/openai-docs/SKILL.md @@ -0,0 +1,161 @@ +--- +name: openai-docs +description: "Use when the user asks how to build with OpenAI products or APIs, asks about Codex itself or choosing Codex surfaces, needs up-to-date official documentation with citations, help choosing the latest model for a use case, or model upgrade and prompt-upgrade guidance; use OpenAI docs MCP tools for non-Codex docs questions, use the Codex manual helper first for broad Codex self-knowledge, and restrict fallback browsing to official OpenAI domains." +--- + + +# OpenAI Docs + +Provide authoritative, current guidance from OpenAI developer docs using the developers.openai.com MCP server. "Docs MCP" means `mcp__openaiDeveloperDocs__search_openai_docs` and `mcp__openaiDeveloperDocs__fetch_openai_doc`; for API reference, schema, parameter, or required-field questions, also use `mcp__openaiDeveloperDocs__get_openapi_spec` when available. Official-domain web search is fallback after those tools are unavailable or unhelpful. Broad Codex questions use the manual helper before Docs MCP. This skill also owns model selection, API model migration, and prompt-upgrade guidance. + +## Workflow Configuration + +### Source Priority + +- For Codex self-knowledge, use the Codex source route below; it owns when to use the manual helper, Docs MCP, or bounded uncertainty. +- For non-Codex OpenAI docs questions, use `mcp__openaiDeveloperDocs__search_openai_docs` to find the most relevant doc pages. +- For non-Codex OpenAI docs questions, fetch the relevant page with `mcp__openaiDeveloperDocs__fetch_openai_doc` before answering. If search is noisy, run a narrower Docs MCP search; when any plausible official OpenAI docs URL is known or found, try fetching that URL through Docs MCP before relying on web-search content. +- For API reference, schema, parameter, or required-field questions, use `mcp__openaiDeveloperDocs__get_openapi_spec` when available to verify the API shape alongside the relevant guide or reference page. +- Use `mcp__openaiDeveloperDocs__list_openai_docs` only when you need to browse or discover non-Codex pages without a clear query. +- For model-selection, "latest model", or default-model questions, fetch `https://developers.openai.com/api/docs/guides/latest-model.md` first. If that is unavailable, load `references/latest-model.md`. +- For model upgrades or prompt upgrades, run `node scripts/resolve-latest-model-info.js` only when the target is latest/current/default or otherwise unspecified; otherwise preserve the explicitly requested target. +- Preserve explicit target requests: if the user names a target model like "migrate to GPT-5.4", keep that requested target even if `latest-model.md` names a newer model. Mention newer guidance only as optional. +- If current remote guidance is needed, fetch both the returned migration and prompting guide URLs directly. If direct fetch fails, use MCP/search fallback; if that also fails, use bundled fallback references and disclose the fallback. + +## OpenAI product snapshots + +1. Apps SDK: Build ChatGPT apps by providing a web component UI and an MCP server that exposes your app's tools to ChatGPT. +2. Responses API: A unified endpoint designed for stateful, multimodal, tool-using interactions in agentic workflows. +3. Chat Completions API: Generate a model response from a list of messages comprising a conversation. +4. Codex: OpenAI's coding agent for software development that can write, understand, review, and debug code. +5. gpt-oss: Open-weight OpenAI reasoning models (gpt-oss-120b and gpt-oss-20b) released under the Apache 2.0 license. +6. Realtime API: Build low-latency, multimodal experiences including natural speech-to-speech conversations. +7. Agents SDK: A toolkit for building agentic apps where a model can use tools and context, hand off to other agents, stream partial results, and keep a full trace. + +## Codex self-knowledge + +Use this path for questions about Codex itself: configuring, extending, operating, troubleshooting, local state, product surfaces, or where Codex behavior should live. A codebase merely mentioning a plugin, skill, hook, MCP server, browser, or automation is not enough. For generic software tasks, answer the software task directly; if asked whether Codex self-knowledge applies, answer that meta question briefly and continue the requested artifact. + +### Source Route + +The Codex manual is the first source for broad Codex synthesis. Treat the manual and Docs MCP as different lanes, not interchangeable official-doc sources. For published-user Codex product answers, the source route is complete: the manual, Docs MCP when this route calls for it, official OpenAI web fallback, and callable capabilities surfaced in the current session when the question is about that capability. Knowledge bases outside developers.openai.com are outside this route for public product answers. + +For broad Codex behavior, setup, customization, skills, plugins, MCP, hooks, `AGENTS.md`, automations, surfaces, local state, or system-map questions: + +1. Reuse a same-thread manual and outline path when it is still fresh. +2. Otherwise run the skill-local helper first in normal writable sessions. Skip it without trying only when the session is explicitly read-only, shell execution is unavailable, or visible policy shows no allowed temp cache. +3. By default, the helper chooses the first usable temp cache dir in this order: `$TMPDIR/openai-docs-cache`, `%TEMP%\openai-docs-cache`, `%TMP%\openai-docs-cache`, `/private/tmp/openai-docs-cache`, then `/tmp/openai-docs-cache`. Workspace-only write access is not enough for this temp cache. +4. Run the helper directly unless you need to override the cache dir. The helper falls back to `curl` when native `fetch` is unavailable or when proxy env vars are present, so no shell-specific proxy prefix is required. Resolve `` to this skill's actual directory; in copied local eval workdirs this is usually `.codex/skills/openai-docs`: + +```bash +node /scripts/fetch-codex-manual.mjs +``` + +If you need to override the cache dir, pass `--cache-dir `. On Windows, the helper checks `%TEMP%` and `%TMP%` automatically; in PowerShell, `$env:TEMP\\openai-docs-cache` is a typical explicit override. + +Treat helper availability as established by explicit read-only/no-shell policy or an actual command result. A guessed sandbox or guessed helper failure is not enough to switch to Docs MCP or web lookup; after an actual helper command failure, continue to the narrowest official next source below. + +The helper verifies freshness, writes `codex-manual.md`, and emits `codex-manual.outline.md`. The outline maps source pages and headings to line ranges; use it to choose the relevant manual section, then read or search targeted manual sections for Codex product facts. Use the skill directory to locate and run the helper; after the helper succeeds, use the returned manual and outline paths as the search scope for Codex product facts and term coverage checks. + +Reuse the same-thread manual and outline paths for follow-up Codex questions. Refresh first when the manual was fetched more than about a day ago, the path is unusable, the path came from another thread or uncertain provenance, or likely-current information is missing and staleness is plausible. + +For questions about whether the manual is current enough to rely on now, run the helper when temp caching is allowed and base the answer on its returned status, manual path, and outline path. + +If the manual resolves a Codex claim, answer from it and stop expanding sources for that claim; continue the user's broader task if the docs lookup was only one dependency. Manual source pages and known anchors are enough citation support for manual-covered material. + +If the helper is skipped because the session is read-only, has no shell execution, or has no allowed temp cache, the next source is Docs MCP: call `mcp__openaiDeveloperDocs__search_openai_docs`, then `mcp__openaiDeveloperDocs__fetch_openai_doc` for a relevant hit before any web fallback. + +If a user names a Codex term or mode that a fresh manual does not use, search the manual for obvious adjacent concepts, then answer that the exact term is not documented and use the closest documented terminology. If the prompt asks how that term maps to Codex behavior, resolve the mapping from adjacent manual sections. If the exact term remains material or likely current after that manual pass, use one narrow Docs MCP search/fetch before bounded uncertainty; otherwise, the source lookup for that terminology or mapping claim is complete. + +Use the narrowest official next source only when the manual is unavailable, the helper fails, temp caching is not allowed, another material claim is missing or likely stale, or the user explicitly needs a page-specific citation. Prefer one specific Docs MCP search and, if it returns a clearly relevant page, one fetch; for unresolved Codex capability names, acronyms, scheduling terms, or exact error text, this Docs MCP step is the next source before web search. After the manual plus any permitted Docs MCP gap-fill, resolve remaining gaps as bounded uncertainty. Use official-domain web fallback only after that Docs MCP path is unavailable or unhelpful. If the claim is still not established, stop with bounded uncertainty. If official docs/manual conflict with a callable capability already surfaced in the current session, state the conflict and prefer verified current-session behavior for that environment. + +For undocumented or private-looking model slugs, product mode labels, entitlement labels, account access paths, or rollout names, answer from current public docs and bounded uncertainty. Those labels are not a reason to leave the public source route. + +For support-style diagnostics, prefer a layer-by-layer answer from the manual over provider-specific web lookups: installed/enabled plugin, bundled app or connector authorization, MCP setup, workspace/admin policy, restart or new-thread expectations, then support or feedback if still unresolved. + +If the source route still does not establish a claim, return bounded uncertainty or route to support, an admin, or product feedback instead of widening the investigation. + +For unresolved product terminology, answer from the manual plus the allowed official next source. If those sources do not establish the term, answer with bounded uncertainty from those sources. + +### Surface Map + +When Codex nouns or durable-instruction surfaces overlap, recommend the smallest surface that matches the scope: + +- Prompt or thread context -> one-off task constraints. +- `AGENTS.md` -> durable repo conventions, commands, verification steps, and review expectations; closer nested files apply under their subtree. +- Project `.codex/config.toml` -> trusted-repo Codex settings such as sandbox, MCP, hooks, model, or reasoning defaults. +- Global config or global guidance -> personal defaults across repos. +- Skill -> reusable task workflow with references or scripts. +- Plugin -> installable bundle with skills plus commands, tools, MCP config, hooks, assets, apps, or marketplace metadata. +- MCP server or app connector -> live external data/actions or authorized private app/workspace data. Use connectors for private Google Docs, Calendar, Slack, GitHub, Notion, and similar data instead of web search or model memory. +- Automation -> scheduled checks, reminders, monitors, or follow-up work; use a thread heartbeat when continuity in an existing thread matters. +- Hook -> lifecycle enforcement around tool calls, commands, or file edits. + +Split mixed-scope requests instead of forcing one answer. Example: "always do X, but only for this PR" defaults to prompt/thread context for the current run; use `AGENTS.md` or project config only if it should persist, hooks only for mechanical enforcement, and automations only for scheduled or follow-up work. + +Use this quick product map when needed: CLI is terminal-first local repo work; IDE extension is editor-attached coding; Codex app is desktop planning, review, and interactive work; cloud/web is hosted parallel/offloaded work; Browser Use/in-app browser is Codex-controlled web testing; Chrome extension uses the user's Chrome profile; Computer Use controls desktop apps and OS UI. Keep `config.toml` defaults, `requirements.toml` constraints, and managed/admin policy separate. + +### Boundaries And Output + +- API key auth does not imply ChatGPT, cloud task, or connector access. For plugin/app/auth failures, check bundle availability, plugin installed/enabled state, connector/app authorization, MCP setup, restart/refresh expectations, workspace policy, and per-surface availability before answering. +- Sandbox or network denials need scoped escalation with a clear justification. Destructive commands, writes outside the workspace, or broad access changes require explicit approval. +- Memory can provide user preference or context, but explicit prompt instructions win and memory is not a source for current external facts. +- For affirmative surface-selection answers, use this shape: recommendation, why, what to avoid, and the manual/source evidence used. +- When page-specific Codex citations are actually needed, these anchors often fit: `concepts/customization#agents-guidance` for `AGENTS.md`, `concepts/customization#skills` for skills, `plugins/build#plugin-structure` for plugins, `concepts/customization#mcp` for MCP, `config-advanced#hooks` for hooks, `app/automations#thread-automations` for thread automations, and `config-reference#configtoml` for config. + +## If MCP server is missing + +If MCP tools fail or no OpenAI docs resources are available: + +1. Run the install command yourself: `codex mcp add openaiDeveloperDocs --url https://developers.openai.com/mcp` +2. If it fails due to permissions/sandboxing, immediately retry the same command with escalated permissions and include a 1-sentence justification for approval. +3. Ask the user to run the install command only if the escalated attempt fails. +4. Ask the user to restart Codex. +5. Re-run the doc search/fetch after restart. + +## Workflow + +1. Clarify whether the request is general docs lookup, model selection, a model-string upgrade, prompt-upgrade guidance, or broader API/provider migration. +2. For Codex self-knowledge requests, follow the Codex self-knowledge source procedure above. +3. For model-selection or upgrade requests, prefer current remote docs over bundled references when the user asks for latest/current/default guidance. + - Fetch `https://developers.openai.com/api/docs/guides/latest-model.md`. + - Find the latest model ID and explicit migration or prompt-guidance links. + - Prefer explicit links from the latest-model page over derived URLs. + - For explicit named-model requests, preserve the requested model target. Mention newer remote guidance only as optional. + - For dynamic latest/current/default upgrades, run `node scripts/resolve-latest-model-info.js`, then fetch both returned guide URLs directly when possible. + - If direct guide fetch fails, use the developer-docs MCP tools or official OpenAI-domain search to find the same guide content. + - If remote docs are unavailable, use bundled fallback references and say that fallback guidance was used. +4. For model upgrades, keep changes narrow: update active OpenAI API model defaults and directly related prompts only when safe. +5. Leave historical docs, examples, eval baselines, fixtures, provider comparisons, provider registries, pricing tables, alias defaults, low-cost fallback paths, and ambiguous older model usage unchanged unless the user explicitly asks to upgrade them. +6. Keep SDK, tooling, IDE, plugin, shell, auth, and provider-environment migrations out of a model-and-prompt upgrade unless the user explicitly asks for them. +7. If an upgrade needs API-surface changes, schema rewiring, tool-handler changes, or implementation work beyond a literal model-string replacement and prompt edits, report it as blocked or confirmation-needed. +8. For general docs lookup, start with a compact, title-like search query of 2-6 essential terms. Do not turn the full user question into a keyword list. Fetch the best page and exact section needed, and answer with concise citations. + +## Reference map + +Read only what you need: + +- `https://developers.openai.com/api/docs/guides/latest-model.md` -> current model-selection and "best/latest/current model" questions. +- `scripts/fetch-codex-manual.mjs` -> current Codex manual fetch, verification, local temp cache, and outline generation. +- `https://developers.openai.com/codex/codex-manual.md` -> current Codex self-knowledge synthesis, including setup, customization, skills, plugins, MCP, hooks, `AGENTS.md`, automations, and surface behavior; normally access it through the helper path and targeted file reads when temp caching is available. +- `references/latest-model.md` -> bundled fallback for model-selection and "best/latest/current model" questions. +- `references/upgrade-guide.md` -> bundled fallback for model upgrade and upgrade-planning requests. +- `references/prompting-guide.md` -> bundled fallback for prompt rewrites and prompt-behavior upgrades. + +## Quality rules + +- Treat OpenAI docs as the source of truth; avoid speculation. +- For Codex self-knowledge, follow the source route above instead of relying on remembered behavior. +- Keep migration changes narrow and behavior-preserving. +- Prefer prompt-only upgrades when possible. +- Avoid inventing pricing, availability, parameters, API changes, or breaking changes. +- Keep quotes short and within policy limits; prefer paraphrase with citations. +- If multiple pages differ, call out the difference and cite both. +- If official docs and verified callable current-session behavior disagree, state the conflict before making broad claims or edits. +- If docs do not cover the user’s need, say so and offer next steps. + +## Tooling notes + +- Use MCP doc tools before web search for OpenAI-related markdown docs. The Codex manual flow is the exception: follow the Codex self-knowledge source procedure for broad Codex synthesis. +- If the MCP server is installed but returns no meaningful results, then use web search as a fallback. +- When falling back to web search, restrict to official OpenAI domains (developers.openai.com, platform.openai.com) and cite sources. diff --git a/.github/skills/openai-docs/agents/openai.yaml b/.github/skills/openai-docs/agents/openai.yaml new file mode 100644 index 00000000..8bbf03c2 --- /dev/null +++ b/.github/skills/openai-docs/agents/openai.yaml @@ -0,0 +1,14 @@ +interface: + display_name: "OpenAI Docs" + short_description: "Reference OpenAI docs, Codex self-knowledge, and model migration guidance" + icon_small: "./assets/openai-small.svg" + icon_large: "./assets/openai.png" + default_prompt: "Use OpenAI Docs for official docs lookup, questions about Codex itself or Codex surfaces, model selection, model migration, and prompt-upgrade work." + +dependencies: + tools: + - type: "mcp" + value: "openaiDeveloperDocs" + description: "OpenAI Developer Docs MCP server" + transport: "streamable_http" + url: "https://developers.openai.com/mcp" diff --git a/.github/skills/openai-docs/assets/openai-small.svg b/.github/skills/openai-docs/assets/openai-small.svg new file mode 100644 index 00000000..1d075dc0 --- /dev/null +++ b/.github/skills/openai-docs/assets/openai-small.svg @@ -0,0 +1,3 @@ + + + diff --git a/.github/skills/openai-docs/assets/openai.png b/.github/skills/openai-docs/assets/openai.png new file mode 100644 index 00000000..e9b9eb80 Binary files /dev/null and b/.github/skills/openai-docs/assets/openai.png differ diff --git a/.github/skills/openai-docs/references/latest-model.md b/.github/skills/openai-docs/references/latest-model.md new file mode 100644 index 00000000..a1ffbfbd --- /dev/null +++ b/.github/skills/openai-docs/references/latest-model.md @@ -0,0 +1,37 @@ +# Latest model guide + +This file is a curated helper. Every recommendation here must be verified against current OpenAI docs before it is repeated to a user. + +## Current model map + +| Model ID | Use for | +| --- | --- | +| `gpt-5.5` | Latest/default text and reasoning model for most new apps, including coding and tool-heavy workflows | +| `gpt-5.5-pro` | Maximum reasoning or quality when latency and cost matter less | +| `gpt-5.4` | Previous default text and reasoning model; use for existing GPT-5.4 integrations | +| `gpt-5.4-mini` | Lower-cost testing and lighter production workflows | +| `gpt-5.4-nano` | High-throughput simple tasks and classification | +| `gpt-5.5` | Explicit no-reasoning text path via `reasoning.effort: none` | +| `gpt-4.1-mini` | Cheaper no-reasoning text | +| `gpt-4.1-nano` | Fastest and cheapest no-reasoning text | +| `gpt-5.3-codex` | Agentic coding, code editing, and tool-heavy coding workflows | +| `gpt-5.1-codex-mini` | Cheaper coding workflows | +| `gpt-image-2` | Best image generation and edit quality | +| `gpt-image-1.5` | Less expensive image generation and edit quality | +| `gpt-image-1-mini` | Cost-optimized image generation | +| `gpt-4o-mini-tts` | Text-to-speech | +| `gpt-4o-mini-transcribe` | Speech-to-text, fast and cost-efficient | +| `gpt-realtime-1.5` | Realtime voice and multimodal sessions | +| `gpt-realtime-mini` | Cheaper realtime sessions | +| `gpt-audio` | Chat Completions audio input and output | +| `gpt-audio-mini` | Cheaper Chat Completions audio workflows | +| `sora-2` | Faster iteration and draft video generation | +| `sora-2-pro` | Higher-quality production video | +| `omni-moderation-latest` | Text and image moderation | +| `text-embedding-3-large` | Higher-quality retrieval embeddings; default in this skill because no best-specific row exists | +| `text-embedding-3-small` | Lower-cost embeddings | + +## Maintenance notes + +- This file will drift unless it is periodically re-verified against current OpenAI docs. +- If this file conflicts with current docs, the docs win. diff --git a/.github/skills/openai-docs/references/prompting-guide.md b/.github/skills/openai-docs/references/prompting-guide.md new file mode 100644 index 00000000..0d9273ce --- /dev/null +++ b/.github/skills/openai-docs/references/prompting-guide.md @@ -0,0 +1,244 @@ +GPT-5.5 works best when prompts define the outcome and leave room for the model to choose an efficient solution path. Compared with earlier models, you can often use shorter, more outcome-oriented prompts: describe what good looks like, what constraints matter, what evidence is available, and what the final answer should contain. + +Avoid carrying over every instruction from an older prompt stack. Legacy prompts often over-specify the process because earlier models needed more help staying on track. With GPT-5.5, that can add noise, narrow the model's search space, or lead to overly mechanical answers. + +For more detail on GPT-5.5 behavior changes, start with the [Using GPT-5.5 guide](/api/docs/guides/latest-model). This guide focuses on prompt changes that follow from those behavior changes. + +The patterns here are starting points. Adapt them to your product surface, tools, evals, and user experience goals. + +## Personality and behavior + +GPT-5.5's default style is efficient, direct, and task-oriented. This is useful for production systems: responses stay focused, behavior is easier to steer, and the model avoids unnecessary conversational padding. + +For customer-facing assistants, support workflows, coaching experiences, and other conversational products, define both personality and collaboration style. + +- **Personality** controls how the assistant sounds: tone, warmth, directness, formality, humor, empathy, and level of polish. +- **Collaboration style** controls how the assistant works: when it asks questions, when it makes assumptions, how proactive it should be, how much context it gives, when it checks work, and how it handles uncertainty or risk. + +Keep both short. Personality instructions should shape the user experience. Collaboration instructions should shape task behavior. Neither should replace clear goals, success criteria, tool rules, or stopping conditions. + +Example personality block for a steady task-focused assistant: + +```text +# Personality +You are a capable collaborator: approachable, steady, and direct. Assume the user is competent and acting in good faith, and respond with patience, respect, and practical helpfulness. + +Prefer making progress over stopping for clarification when the request is already clear enough to attempt. Use context and reasonable assumptions to move forward. Ask for clarification only when the missing information would materially change the answer or create meaningful risk, and keep any question narrow. + +Stay concise without becoming curt. Give enough context for the user to understand and trust the answer, then stop. Use examples, comparisons, or simple analogies when they make the point easier to grasp. When correcting the user or disagreeing, be candid but constructive. When an error is pointed out, acknowledge it plainly and focus on fixing it. + +Match the user's tone within professional bounds. Avoid emojis and profanity by default, unless the user explicitly asks for that style or has clearly established it as appropriate for the conversation. +``` + +Example personality block for an expressive collaborative assistant: + +```text +# Personality +Adopt a vivid conversational presence: intelligent, curious, playful when appropriate, and attentive to the user's thinking. Ask good questions when the problem is blurry, then become decisive once there is enough context. + +Be warm, collaborative, and polished. Conversation should feel easy and alive, but not chatty for its own sake. Offer a real point of view rather than merely mirroring the user, while staying responsive to their goals and constraints. + +Be thoughtful and grounded when the task calls for synthesis or advice. State a clear recommendation when you have enough context, explain important tradeoffs, and name uncertainty without becoming evasive. +``` + +For more expressive products, add warmth, curiosity, humor, or point of view explicitly, but keep the block short. Use personality to shape the experience, not to compensate for unclear goals or missing task instructions. + +## Improve time to first visible token with a preamble + +In streaming applications, users notice how long it takes before the first visible response appears. GPT-5.5 may spend time reasoning, planning, or preparing tool calls before emitting visible text. + +For longer or tool-heavy tasks, prompt the model to start with a short preamble: a brief visible update that acknowledges the request and states the first step. This can improve perceived responsiveness without changing the underlying task. + +Use this pattern when the task may take more than one step, require tool calls, or involve a long-running agent workflow. + +```text +Before any tool calls for a multi-step task, send a short user-visible update that acknowledges the request and states the first step. Keep it to one or two sentences. +``` + +For coding agents that expose separate message phases, you can be more explicit: + +```text +You must always start with an intermediary update before any content in the analysis channel if the task will require calling tools. The user update should acknowledge the request and explain your first step. +``` + +## Outcome-first prompts and stopping conditions + +GPT-5.5 is strongest when the prompt defines the target outcome, success criteria, constraints, and available context, then lets the model choose the path. + +For many tasks, describe the destination rather than every step. This gives the model room to choose the right search, tool, or reasoning strategy for the task. + +Prefer this: + +```text +Resolve the customer's issue end to end. + +Success means: +- the eligibility decision is made from the available policy and account data +- any allowed action is completed before responding +- the final answer includes completed_actions, customer_message, and blockers +- if evidence is missing, ask for the smallest missing field +``` + +**Avoid unnecessary absolute rules.** Older prompts often use strict instructions like `ALWAYS`, `NEVER`, `must`, and `only` to control model behavior. Use those words for true invariants, such as safety rules, required output fields, or actions that should never happen. For judgment calls, such as when to search, ask for clarification, use a tool, or keep iterating, prefer decision rules instead. + +Avoid this style of instruction unless every step is truly required: + +```text +First inspect A, then inspect B, then compare every field, then think through +all possible exceptions, then decide which tool to call, then call the tool, +then explain the entire process to the user. +``` + +Add explicit stopping conditions: + +```text +Resolve the user query in the fewest useful tool loops, but do not let loop minimization outrank correctness, accessible fallback evidence, calculations, or required citation tags for factual claims. + +After each result, ask: "Can I answer the user's core request now with useful evidence and citations for the factual claims?" If yes, answer. +``` + +Define missing-evidence behavior: + +```text +Use the minimum evidence sufficient to answer correctly, cite it precisely, then stop. +``` + +## Formatting + +GPT-5.5 is highly steerable on output format and structure. Use that control when it improves comprehension or product fit. + +Set `text.verbosity`, describe the expected output shape, and reserve heavier structure for cases where it improves comprehension or your product UI needs a stable artifact. The API default for `text.verbosity` is `medium`; use `low` when you prefer shorter, more concise responses. + +Plain conversational formatting: + +```text +Let formatting serve comprehension. Use plain paragraphs as the default format for normal conversation, explanations, reports, documentation, and technical writeups. Keep the presentation clean and readable without making the structure feel heavier than the content. + +Use headers, bold text, bullets, and numbered lists sparingly. Reach for them when the user requests them, when the answer needs clear comparison or ranking, or when the information would be harder to scan as prose. Otherwise, favor short paragraphs and natural transitions. + +Respect formatting preferences from the user. If they ask for a terse answer, minimal formatting, no bullets, no headers, or a specific structure, follow that preference unless there is a strong reason not to. +``` + +Add explicit audience and length guidance: + +```text +Write for a senior business audience. Keep the answer under 400 words. Use short paragraphs and only include bullets when they improve scannability. Prioritize the conclusion first, then the reasoning, then caveats. +``` + +For editing, rewriting, summaries, or customer-facing messages, tell the model what to preserve before asking it to improve style. This pattern is useful when you want polish without expansion. + +```text +Preserve the requested artifact, length, structure, and genre first. Quietly improve clarity, flow, and correctness. Do not add new claims, extra sections, or a more promotional tone unless explicitly requested. +``` + +## Grounding, citations, and retrieval budgets + +For grounded answers, citation behavior should be part of the prompt. Define what needs support, what counts as enough evidence, and how the model should behave when evidence is missing. Absence of evidence shouldn't automatically become a factual "no." For more details and examples, see the [citation formatting guide](/api/docs/guides/citation-formatting). + +### Add an explicit retrieval budget + +Retrieval budgets are stopping rules for search. They tell the model when enough evidence is enough. + +```text +For ordinary Q&A, start with one broad search using short, discriminative keywords. If the top results contain enough citable support for the core request, answer from those results instead of searching again. + +Make another retrieval call only when: +- The top results do not answer the core question. +- A required fact, parameter, owner, date, ID, or source is missing. +- The user asked for exhaustive coverage, a comparison, or a comprehensive list. +- A specific document, URL, email, meeting, record, or code artifact must be read. +- The answer would otherwise contain an important unsupported factual claim. + +Do not search again to improve phrasing, add examples, cite nonessential details, or support wording that can safely be made more generic. +``` + +## Creative drafting guardrails + +For drafting tasks, tell the model which claims must come from sources and which parts may be creatively written. This is especially important for slides, launch copy, customer summaries, talk tracks, leadership blurbs, and narrative framing. + +```text +For creative or generative requests such as slides, leadership blurbs, outbound copy, summaries for sharing, talk tracks, or narrative framing, distinguish source-backed facts from creative wording. + +- Use retrieved or provided facts for concrete product, customer, metric, roadmap, date, capability, and competitive claims, and cite those claims. +- Do not invent specific names, first-party data claims, metrics, roadmap status, customer outcomes, or product capabilities to make the draft sound stronger. +- If there is little or no citable support, write a useful generic draft with placeholders or clearly labeled assumptions rather than unsupported specifics. +``` + +## Frontend engineering and visual taste + +For frontend work, refer to the [example instructions](/api/docs/guides/frontend-prompt) for practical ways to steer UI quality. They cover product and user context, design-system alignment, first-screen usability, familiar controls, expected states, responsive behavior, and common generated-UI defaults to avoid, such as generic heroes, nested cards, decorative gradients, visible instructional text, and broken layouts. + +## Prompt the model to check its work + +Give GPT-5.5 access to tools that let it check outputs when validation is possible. + +For coding agents, ask for concrete validation commands: + +```text +After making changes, run the most relevant validation available: +- targeted unit tests for changed behavior +- type checks or lint checks when applicable +- build checks for affected packages +- a minimal smoke test when full validation is too expensive + +If validation cannot be run, explain why and describe the next best check. +``` + +For visual artifacts, ask for inspection after rendering: + +```text +Render the artifact before finalizing. Inspect the rendered output for layout, clipping, spacing, missing content, and visual consistency. Revise until the rendered output matches the requirements. +``` + +For engineering and planning tasks, make implementation plans traceable: + +```text +For implementation plans, include: +- requirements and where each is addressed +- named resources, files, APIs, or systems involved +- state transitions or data flow where relevant +- validation commands or checks +- failure behavior +- privacy and security considerations +- open questions that materially affect implementation +``` + +## Phase parameter + +Starting with GPT-5.4, long-running or tool-heavy Responses workflows can use assistant-item `phase` values to distinguish intermediate updates from final answers. GPT-5.5 uses the same pattern. + +If you use `previous_response_id`, the API preserves prior assistant state automatically. If your application manually replays assistant output items into the next request, preserve each original `phase` value and pass it back unchanged. This matters most when a response includes preambles, repeated tool calls, or a final answer after intermediate assistant updates. + +```text +If manually replaying assistant items: +- Preserve assistant `phase` values exactly. +- Use `phase: "commentary"` for intermediate user-visible updates. +- Use `phase: "final_answer"` for the completed answer. +- Do not add `phase` to user messages. +``` + +## Suggested prompt structure + +Use this structure as a starting point for complex prompts. Keep each section short. Add detail only where it changes behavior. + +```text +Role: [1-2 sentences defining the model's function, context, and job] + +# Personality +[tone, demeanor, and collaboration style] + +# Goal +[user-visible outcome] + +# Success criteria +[what must be true before the final answer] + +# Constraints +[policy, safety, business, evidence, and side-effect limits] + +# Output +[sections, length, and tone] + +# Stop rules +[when to retry, fallback, abstain, ask, or stop] +``` diff --git a/.github/skills/openai-docs/references/upgrade-guide.md b/.github/skills/openai-docs/references/upgrade-guide.md new file mode 100644 index 00000000..b29f137b --- /dev/null +++ b/.github/skills/openai-docs/references/upgrade-guide.md @@ -0,0 +1,181 @@ +# Upgrading to GPT-5.5 + +Use this guide when the user explicitly asks to upgrade an existing integration to GPT-5.5. Pair it with current OpenAI docs lookups. The default target string is `gpt-5.5`. + +## Freshness check + +Before applying this bundled guide for a latest/current/default model upgrade, run `node scripts/resolve-latest-model-info.js` from the OpenAI Docs skill directory. + +- If the command returns `modelSlug: "gpt-5p5"`, continue with this bundled guide and use `references/prompting-guide.md` when prompt updates are needed. +- If the command returns a different `modelSlug`, fetch both the returned `migrationGuideUrl` and `promptingGuideUrl` and use them as the current source of truth instead of the bundled references. +- If the command fails, metadata is missing, or either remote guide cannot be fetched, continue with bundled fallback references and say the remote freshness check was unavailable. +- If the user explicitly named a target model, preserve that target and use current docs only to check compatibility or caveats. + +## Upgrade posture + +Upgrade with the narrowest safe change set: + +- replace the model string first +- update only the prompts that are directly tied to that model usage +- do not automatically upgrade older or ambiguous model usages that may be intentionally pinned, such as historical docs, examples, tests, eval baselines, comparison code, or low-cost fallback/routing paths. Unless the user explicitly asks to upgrade all model usage, leave those sites unchanged and list them as confirmation-needed +- prefer prompt-only upgrades when possible +- if the upgrade would require API-surface changes, parameter rewrites, tool rewiring, provider migration, or broader code edits, mark it as blocked instead of stretching the scope + +## Upgrade workflow + +1. Inventory current model usage. + - Search for model strings, client calls, and prompt-bearing files. + - Include inline prompts, prompt templates, YAML or JSON configs, Markdown docs, and saved prompts when they are clearly tied to a model usage site. +2. Pair each model usage with its prompt surface. + - Prefer the closest prompt surface first: inline system or developer text, then adjacent prompt files, then shared templates. + - If you cannot confidently tie a prompt to the model usage, say so instead of guessing. +3. Classify the source model family. + - Common buckets: GPT-5.4, GPT-5.3-Codex or GPT-5.2-Codex, earlier GPT-5.x, GPT-4o or GPT-4.1, reasoning models such as o1 or o3 or o4-mini, third-party model, or mixed and unclear. +4. Decide the upgrade class. + - `model string only` + - `model string + light prompt rewrite` + - `blocked without code changes` +5. Run the compatibility gate. + - Check whether the current integration can accept `gpt-5.5` without API-surface changes or implementation changes. + - Check whether structured outputs, tool schemas, function names, and downstream parsers can remain unchanged. + - For long-running Responses or tool-heavy agents, check whether `phase` is already preserved or round-tripped when the host replays assistant items or uses preambles. + - If compatibility depends on code changes, return `blocked`. + - If compatibility is unclear, return `unknown` rather than improvising. +6. Apply the upgrade when it is in scope. + - Default replacement string: `gpt-5.5`. + - Keep the intervention small and behavior-preserving. + - Start from the current reasoning effort when it is visible unless there is a measured reason to change it. + - For in-scope changes, update the model string and directly related prompts. + - For blocked or unknown changes, do not edit; report the blocker or uncertainty. +7. Summarize the result. + - `Current model usage` + - `Model-string updates` + - `Reasoning-effort handling` + - `Prompt updates` + - `Structured output and formatting assessment` + - `Tool-use assessment` when the flow uses tools, retrieval, or terminal actions + - `Phase assessment` when the flow is long-running, replayed, or tool-heavy + - `Compatibility check` + - `Validation performed` + +Output rule: + +- For each usage site, state the starting reasoning-effort recommendation. +- If the repo exposes the current reasoning setting, recommend preserving it first unless current OpenAI docs say otherwise. +- If the repo does not expose the current setting, recommend not adding one unless current OpenAI docs require it. + +## Upgrade outcomes + +### `model string only` + +Choose this when: + +- the source model is GPT-5.4 +- the existing prompts are already short, explicit, and task-bounded +- the workflow does not rely on strict output formats, tool-call behavior, batch completeness, or long-horizon execution that should be validated after the upgrade +- there are no obvious compatibility blockers + +Default action: + +- replace the model string with `gpt-5.5` +- preserve the current reasoning effort +- keep prompts unchanged +- validate behavior with existing tests, realistic spot checks, or an existing eval suite when one is already available + +### `model string + light prompt rewrite` + +Choose this when: + +- the task needs stronger completeness, citation discipline, verification, or dependency handling +- the upgraded model becomes too verbose, too dense, or hard to scan unless formatting is constrained +- the workflow has strict output shape requirements and lacks an explicit format contract, schema, or parser validation +- the workflow is research-heavy and needs stronger handling of sparse or empty retrieval results +- the workflow is coding-oriented, terminal-based, tool-heavy, or multi-agent, but the existing API surface and tool definitions can remain unchanged + +Default action: + +- replace the model string with `gpt-5.5` +- preserve the current reasoning effort for the first pass +- make only the smallest prompt edits needed for the observed workflow risk +- read the [GPT-5.5 prompting guide](/api/docs/guides/prompt-guidance?model=gpt-5.5) to choose the smallest prompt changes that recover or improve behavior +- avoid broad prompt cleanup unrelated to the upgrade +- for research workflows, add citation rules, retrieval budgets, missing-evidence behavior, and validation guidance from the prompting guide +- for dependency-aware or tool-heavy workflows, add prerequisite checks, missing-context handling, explicit tool budgets, stop conditions, and validation guidance +- for coding or terminal workflows, add repo-specific constraints, acceptance criteria, and concrete validation commands +- for multi-agent support or triage workflows, add task ownership, handoff, completeness, and stopping criteria +- for long-running Responses agents with preambles or multiple assistant messages, explicitly review whether `phase` is already handled; if adding or preserving `phase` would require code edits, mark the path as `blocked` +- do not classify a coding or tool-using Responses workflow as `blocked` just because the visible snippet is minimal; prefer `model string + light prompt rewrite` unless the repo clearly shows that a safe GPT-5.5 path would require host-side code changes + +### `blocked` + +Choose this when: + +- the upgrade appears to require API-surface changes +- the upgrade appears to require parameter rewrites or reasoning-setting changes that are not exposed outside implementation code +- the upgrade would require changing tool definitions, tool handler wiring, or schema contracts +- the user is asking for a tooling, IDE, plugin, shell, or environment migration rather than a model and prompt migration +- the integration depends on provider-specific APIs that do not map to the current OpenAI API surface without implementation work +- you cannot confidently identify the prompt surface tied to the model usage + +Default action: + +- do not improvise a broader upgrade +- report the blocker and explain that the fix is out of scope for this guide +- if useful, describe the smallest follow-up implementation task that would unblock the migration + +## Compatibility checklist + +Before applying or recommending a model-and-prompt-only upgrade, check: + +1. Can the current host accept the `gpt-5.5` model string without changing client code or API surface? +2. Are the related prompts identifiable and editable? +3. Does the host depend on behavior that likely needs API-surface changes, parameter rewrites, provider migration, or tool rewiring? +4. Would the likely fix be prompt-only, or would it need implementation changes? +5. Is the prompt surface close enough to the model usage that you can make a targeted change instead of a broad cleanup? +6. Do strict structured outputs, schemas, or downstream parsers still have an explicit contract? +7. For long-running Responses or tool-heavy agents, is `phase` already preserved if the host relies on preambles, replayed assistant items, or multiple assistant messages? +8. Are latency, token, or price assumptions validated by tests, realistic spot checks, or an existing eval suite rather than inferred from general model positioning? + +If item 1 is no, items 3 through 4 point to implementation work, or item 7 is no and the fix needs code changes, return `blocked`. + +If item 2 is no, return `unknown` unless the user can point to the prompt location. + +Important: + +- Existing use of tools, agents, or multiple usage sites is not by itself a blocker. +- If the current host can keep the same API surface and the same tool definitions, prefer `model string + light prompt rewrite` over `blocked`. +- Reserve `blocked` for cases that truly require implementation changes, not cases that only need stronger prompt steering. +- Do not claim token savings without task-level validation. + +## Scope boundaries + +This guide may: + +- update or recommend updated model strings +- update or recommend updated prompts +- inspect code and prompt files to understand where those changes belong +- inspect whether existing Responses flows already preserve `phase` +- flag compatibility blockers +- propose validation with existing tests, realistic spot checks, or existing eval suites + +This guide may not: + +- move Chat Completions code to Responses +- move Responses code to another API surface +- migrate SDKs, APIs, IDE configuration, shell hooks, plugins, or provider-specific tooling +- rewrite parameter shapes +- change tool definitions or tool-call handling +- change structured-output wiring +- add or retrofit `phase` handling in implementation code +- edit business logic, orchestration logic, SDK usage, IDE configuration, shell hooks, or plugin integration behavior except for model-string replacements and directly related prompt edits + +If a safe GPT-5.5 upgrade requires any of those changes, mark the path as blocked and out of scope. + +## Validation plan + +- Validate each upgraded usage site with existing tests, realistic spot checks, or an existing eval suite when one is already available. +- Compare against the current GPT-5.4 baseline when available. +- Check task success, retry count, tool-call count, total tokens, latency, output shape, and user-visible quality. +- For specialized workflows, validate the contract that matters most instead of judging only general output quality. +- If prompt edits were added, confirm each block is doing real work instead of adding noise. +- If the workflow has downstream impact, add a lightweight verification pass before finalization. diff --git a/.github/skills/openai-docs/scripts/fetch-codex-manual.mjs b/.github/skills/openai-docs/scripts/fetch-codex-manual.mjs new file mode 100644 index 00000000..b2605520 --- /dev/null +++ b/.github/skills/openai-docs/scripts/fetch-codex-manual.mjs @@ -0,0 +1,598 @@ +#!/usr/bin/env node +import { + access, + mkdir, + readFile, + rename, + rm, + stat, + writeFile, +} from "node:fs/promises"; +import { constants as fsConstants } from "node:fs"; +import { execFile } from "node:child_process"; +import { createHash } from "node:crypto"; +import path from "node:path"; +import process from "node:process"; +import { pathToFileURL } from "node:url"; +import { inspect, promisify } from "node:util"; + +const DEFAULT_MANUAL_URL = "https://developers.openai.com/codex/codex-manual.md"; +const DEFAULT_CACHE_DIR_NAME = "openai-docs-cache"; +const CACHE_FILE_NAME = "codex-manual.md"; +const OUTLINE_FILE_NAME = "codex-manual.outline.md"; +const HASH_HEADER = "x-content-sha256"; +const USER_AGENT = "codex-openai-docs"; +const execFileAsync = promisify(execFile); + +class ManualFetchError extends Error { + constructor(message, options) { + super(message, options); + this.name = "ManualFetchError"; + } +} + +const sha256 = (value) => createHash("sha256").update(value).digest("hex"); + +const withTimeout = async (promiseFactory, timeoutMs) => { + const controller = new AbortController(); + const timeout = setTimeout(() => controller.abort(), timeoutMs); + try { + return await promiseFactory(controller.signal); + } finally { + clearTimeout(timeout); + } +}; + +const proxyConfigured = () => + process.env.HTTP_PROXY || + process.env.HTTPS_PROXY || + process.env.http_proxy || + process.env.https_proxy; + +const responseHeaders = (headers) => ({ + get(name) { + return headers.get(name.toLowerCase()) ?? null; + }, +}); + +const makeResponse = ({ body, headers, status }) => ({ + headers: responseHeaders(headers), + ok: status >= 200 && status < 300, + status, + async text() { + return body; + }, +}); + +const parseCurlHeaders = (rawHeaders) => { + const normalized = rawHeaders.replace(/\r\n/g, "\n").trim(); + const blocks = normalized.split(/\n\n+/).filter(Boolean); + const headerBlock = [...blocks] + .reverse() + .find((block) => block.startsWith("HTTP/")); + + if (!headerBlock) { + throw new ManualFetchError("curl did not return HTTP response headers."); + } + + const [statusLine, ...lines] = headerBlock.split("\n"); + const statusMatch = /^HTTP\/\S+\s+(\d{3})/.exec(statusLine); + if (!statusMatch) { + throw new ManualFetchError( + `Could not parse HTTP status from curl response: ${statusLine}` + ); + } + + const headers = new Map(); + lines.forEach((line) => { + const separator = line.indexOf(":"); + if (separator === -1) return; + const name = line.slice(0, separator).trim().toLowerCase(); + const value = line.slice(separator + 1).trim(); + headers.set(name, value); + }); + + return { + headers, + status: Number(statusMatch[1]), + }; +}; + +const tempFilePath = (cacheDir, suffix) => + path.join( + cacheDir, + `.fetch-codex-manual-${process.pid}-${Date.now()}-${Math.random() + .toString(16) + .slice(2)}${suffix}` + ); + +const requestManualWithCurl = async (url, { cacheDir, method, timeoutMs }) => { + const headerPath = tempFilePath(cacheDir, ".headers"); + const bodyPath = tempFilePath(cacheDir, ".body"); + const curlNames = + process.platform === "win32" ? ["curl.exe", "curl"] : ["curl"]; + const args = [ + "--silent", + "--show-error", + "--location", + "--dump-header", + headerPath, + "--output", + bodyPath, + "--user-agent", + USER_AGENT, + "--max-time", + String(Math.max(1, Math.ceil(timeoutMs / 1000))), + ]; + + if (method === "HEAD") { + args.push("--head"); + } else { + args.push("--request", method); + } + args.push(url); + + let lastError; + for (const curlName of curlNames) { + try { + await execFileAsync(curlName, args, { windowsHide: true }); + const [rawHeaders, body] = await Promise.all([ + readFile(headerPath, "utf8"), + readFile(bodyPath, "utf8"), + ]); + const { headers, status } = parseCurlHeaders(rawHeaders); + return makeResponse({ body, headers, status }); + } catch (error) { + lastError = error; + if (error?.code !== "ENOENT") break; + } finally { + await Promise.all([ + rm(headerPath, { force: true }), + rm(bodyPath, { force: true }), + ]); + } + } + + if (lastError?.code === "ENOENT") { + throw new ManualFetchError("curl is unavailable in this environment.", { + cause: lastError, + }); + } + throw new ManualFetchError(`${method} ${url} could not be fetched.`, { + cause: lastError, + }); +}; + +const requestManualWithFetch = async (url, { method, timeoutMs }) => { + if (typeof fetch !== "function") { + throw new ManualFetchError( + "Native fetch is unavailable in this Node runtime." + ); + } + + return withTimeout( + (signal) => + fetch(url, { + method, + headers: { "User-Agent": USER_AGENT }, + signal, + }), + timeoutMs + ); +}; + +const requestManual = async (url, { cacheDir, method, timeoutMs }) => { + const preferCurl = Boolean(proxyConfigured()) || typeof fetch !== "function"; + const transports = preferCurl + ? [ + () => requestManualWithCurl(url, { cacheDir, method, timeoutMs }), + () => requestManualWithFetch(url, { method, timeoutMs }), + ] + : [ + () => requestManualWithFetch(url, { method, timeoutMs }), + () => requestManualWithCurl(url, { cacheDir, method, timeoutMs }), + ]; + + let lastError; + for (const transport of transports) { + try { + const response = await transport(); + if (!response.ok) { + throw new ManualFetchError( + `${method} ${url} failed with HTTP ${response.status}.` + ); + } + return response; + } catch (error) { + lastError = error; + } + } + + throw new ManualFetchError(`${method} ${url} could not be fetched.`, { + cause: lastError, + }); +}; + +const readHeaderSha = (response) => { + const value = response.headers.get(HASH_HEADER); + if (!value || !/^[a-f0-9]{64}$/i.test(value)) { + throw new ManualFetchError(`Manual response is missing ${HASH_HEADER}.`); + } + return value.toLowerCase(); +}; + +const nearestExistingParent = async (target) => { + let current = target; + while (true) { + try { + const info = await stat(current); + return info.isDirectory() ? current : null; + } catch (error) { + if (error?.code !== "ENOENT") return null; + } + + const parent = path.dirname(current); + if (parent === current) return null; + current = parent; + } +}; + +const usableCacheDir = async (cacheDir) => { + if (!cacheDir) return null; + const resolved = path.resolve(cacheDir); + + try { + const info = await stat(resolved); + if (!info.isDirectory()) return null; + } catch (error) { + if (error?.code !== "ENOENT") return null; + } + + const parent = await nearestExistingParent(resolved); + if (!parent) return null; + + try { + await access(parent, fsConstants.W_OK | fsConstants.X_OK); + } catch { + return null; + } + + return resolved; +}; + +const defaultCacheDirCandidates = () => { + const candidates = []; + const seen = new Set(); + const pushCandidate = (candidate) => { + if (!candidate || seen.has(candidate)) return; + seen.add(candidate); + candidates.push(candidate); + }; + + [process.env.TMPDIR, process.env.TEMP, process.env.TMP].forEach((baseDir) => { + if (baseDir) { + pushCandidate(path.join(baseDir, DEFAULT_CACHE_DIR_NAME)); + } + }); + + if (process.platform !== "win32") { + pushCandidate(`/private/tmp/${DEFAULT_CACHE_DIR_NAME}`); + pushCandidate(`/tmp/${DEFAULT_CACHE_DIR_NAME}`); + } + + return candidates; +}; + +const resolveCacheDir = async (cacheDir) => { + if (cacheDir) { + return usableCacheDir(cacheDir); + } + + for (const candidate of defaultCacheDirCandidates()) { + const usable = await usableCacheDir(candidate); + if (usable) return usable; + } + + return null; +}; + +const cacheFilePath = (cacheDir) => path.join(cacheDir, CACHE_FILE_NAME); + +const outlineFilePath = (cacheDir) => path.join(cacheDir, OUTLINE_FILE_NAME); + +const manualLines = (manual) => { + const lines = manual.replace(/\r\n/g, "\n").split("\n"); + if (lines[lines.length - 1] === "") lines.pop(); + return lines; +}; + +const sectionTitle = (rawTitle) => + rawTitle.replace(/\s+#+\s*$/, "").replace(/\s+/g, " ").trim(); + +const buildOutline = (manual) => { + const lines = manualLines(manual); + const headings = []; + let inFence = false; + + lines.forEach((line, index) => { + if (/^\s*(```|~~~)/.test(line)) { + inFence = !inFence; + return; + } + if (inFence) return; + + const match = /^(#{1,6})\s+(.+?)\s*$/.exec(line); + if (!match) return; + + const level = match[1].length; + if (level < 2 || level > 3) return; + + headings.push({ + level, + title: sectionTitle(match[2]), + startLine: index + 1, + endLine: lines.length, + }); + }); + + for (let index = 0; index < headings.length; index += 1) { + const heading = headings[index]; + const nextPeer = headings + .slice(index + 1) + .find((candidate) => candidate.level <= heading.level); + if (nextPeer) { + heading.endLine = nextPeer.startLine - 1; + } + } + + if (headings.length === 0) { + return { + headingCount: 0, + lineCount: lines.length, + text: "No markdown headings found.", + }; + } + + const minLevel = Math.min(...headings.map((heading) => heading.level)); + return { + headingCount: headings.length, + lineCount: lines.length, + text: headings + .map((heading) => { + const indent = " ".repeat(heading.level - minLevel); + return `${indent}- ${heading.title} (lines ${heading.startLine}-${heading.endLine})`; + }) + .join("\n"), + }; +}; + +const outlineMarkdown = (outline) => `# Codex Manual Outline\n\n${outline.text}\n`; + +const manualStatusLine = (status) => + status.cacheStatus === "hit" + ? "Manual status: local manual was already current." + : "Manual status: local manual was updated."; + +const formatResult = ({ status, outlineText }) => + [ + `Manual path: ${status.manualPath}`, + `Outline path: ${status.outlinePath}`, + manualStatusLine(status), + "", + outlineText, + ].join("\n"); + +const readCachedManual = async (cacheDir, expectedSha256) => { + try { + const manual = await readFile(cacheFilePath(cacheDir), "utf8"); + return sha256(manual) === expectedSha256 ? manual : null; + } catch { + return null; + } +}; + +const writeCachedManual = async (cacheDir, manual) => { + await mkdir(cacheDir, { recursive: true }); + const tmpPath = tempFilePath(cacheDir, `.${CACHE_FILE_NAME}.tmp`); + await writeFile(tmpPath, manual, "utf8"); + await rename(tmpPath, cacheFilePath(cacheDir)); +}; + +const writeOutline = async (cacheDir, outlineText) => { + await mkdir(cacheDir, { recursive: true }); + const tmpPath = tempFilePath(cacheDir, `.${OUTLINE_FILE_NAME}.tmp`); + await writeFile(tmpPath, outlineText, "utf8"); + await rename(tmpPath, outlineFilePath(cacheDir)); +}; + +const fetchCodexManual = async ({ + manualUrl = DEFAULT_MANUAL_URL, + cacheDir, + timeoutMs = 30000, +} = {}) => { + const resolvedCacheDir = await resolveCacheDir(cacheDir); + if (!resolvedCacheDir) { + throw new ManualFetchError( + "Manual cache directory is unavailable; pass --cache-dir to override or use OpenAI Docs MCP fallback." + ); + } + await mkdir(resolvedCacheDir, { recursive: true }); + + const headResponse = await requestManual(manualUrl, { + cacheDir: resolvedCacheDir, + method: "HEAD", + timeoutMs, + }); + const expectedSha256 = readHeaderSha(headResponse); + const manualPath = cacheFilePath(resolvedCacheDir); + const outlinePath = outlineFilePath(resolvedCacheDir); + const checkedAt = new Date().toISOString(); + + const cachedManual = await readCachedManual(resolvedCacheDir, expectedSha256); + if (cachedManual !== null) { + const outline = buildOutline(cachedManual); + const outlineText = outlineMarkdown(outline); + await writeOutline(resolvedCacheDir, outlineText); + + return { + outlineText, + status: { + manualUrl, + headerSha256: expectedSha256, + fetchedManualSha256: expectedSha256, + manualHashMatches: true, + cacheStatus: "hit", + cacheDir: resolvedCacheDir, + manualPath, + outlinePath, + checkedAt, + lineCount: outline.lineCount, + headingCount: outline.headingCount, + }, + }; + } + + const getResponse = await requestManual(manualUrl, { + cacheDir: resolvedCacheDir, + method: "GET", + timeoutMs, + }); + const getHeaderSha256 = readHeaderSha(getResponse); + if (getHeaderSha256 !== expectedSha256) { + throw new ManualFetchError( + `${HASH_HEADER} changed between HEAD and GET for ${manualUrl}.` + ); + } + + const manualText = await getResponse.text(); + const actualSha256 = sha256(manualText); + const manualHashMatches = actualSha256 === expectedSha256; + if (!manualHashMatches) { + throw new ManualFetchError( + `${HASH_HEADER} did not match the fetched manual body for ${manualUrl}.` + ); + } + + await writeCachedManual(resolvedCacheDir, manualText); + const outline = buildOutline(manualText); + const outlineText = outlineMarkdown(outline); + await writeOutline(resolvedCacheDir, outlineText); + + return { + outlineText, + status: { + manualUrl, + headerSha256: expectedSha256, + fetchedManualSha256: actualSha256, + manualHashMatches, + cacheStatus: "updated", + cacheDir: resolvedCacheDir, + manualPath, + outlinePath, + checkedAt, + lineCount: outline.lineCount, + headingCount: outline.headingCount, + }, + }; +}; + +const parseArgs = (argv) => { + const args = { + manualUrl: DEFAULT_MANUAL_URL, + cacheDir: undefined, + timeoutMs: 30000, + statusJson: false, + }; + + for (let index = 0; index < argv.length; index += 1) { + const arg = argv[index]; + if (arg === "--manual-url") { + args.manualUrl = argv[++index]; + } else if (arg === "--cache-dir") { + args.cacheDir = argv[++index]; + } else if (arg === "--timeout-ms") { + args.timeoutMs = Number(argv[++index]); + } else if (arg === "--status-json") { + args.statusJson = true; + } else { + throw new ManualFetchError(`Unknown argument: ${arg}`); + } + } + + if (!args.manualUrl) { + throw new ManualFetchError("--manual-url cannot be empty."); + } + if (!Number.isFinite(args.timeoutMs) || args.timeoutMs <= 0) { + throw new ManualFetchError("--timeout-ms must be a positive number."); + } + + return args; +}; + +const main = async () => { + const args = parseArgs(process.argv.slice(2)); + const { outlineText, status } = await fetchCodexManual(args); + + process.stdout.write(formatResult({ status, outlineText })); + + if (args.statusJson) { + console.error(JSON.stringify(status)); + } +}; + +const envProxyHint = () => { + if (proxyConfigured()) { + return "Hint: proxy env vars are present. This helper prefers `curl` in proxied sessions; if requests still fail, verify `curl` is installed and the proxy configuration is valid."; + } + if (typeof fetch !== "function") { + return "Hint: native fetch is unavailable in this Node runtime. Install `curl` or use a newer Node version to fetch the manual."; + } + if (process.platform === "win32") { + return "Hint: on Windows, pass a cache dir under `%TEMP%` or `%TMP%`."; + } + return null; +}; + +const formatErrorDetails = (error) => { + const details = inspect(error, { + breakLength: 120, + colors: false, + compact: false, + depth: 8, + }); + if (!error?.cause) { + return details; + } + + return `${details}\n\nCause:\n${inspect(error.cause, { + breakLength: 120, + colors: false, + compact: false, + depth: 8, + })}`; +}; + +const isCliEntrypoint = () => { + const entrypoint = process.argv[1]; + if (!entrypoint) { + return false; + } + + return pathToFileURL(entrypoint).href === import.meta.url; +}; + +if (isCliEntrypoint()) { + main().catch((error) => { + console.error(`Error: ${error.message}`); + const hint = envProxyHint(); + if (hint) { + console.error(hint); + } + console.error(""); + console.error("Details:"); + console.error(formatErrorDetails(error)); + process.exitCode = 1; + }); +} + +export { DEFAULT_MANUAL_URL, fetchCodexManual }; diff --git a/.github/skills/openai-docs/scripts/resolve-latest-model-info.js b/.github/skills/openai-docs/scripts/resolve-latest-model-info.js new file mode 100755 index 00000000..1bd16ac9 --- /dev/null +++ b/.github/skills/openai-docs/scripts/resolve-latest-model-info.js @@ -0,0 +1,147 @@ +#!/usr/bin/env node + +const fs = require("node:fs/promises"); +const path = require("node:path"); + +const DEFAULT_URL = + "https://developers.openai.com/api/docs/guides/latest-model.md"; +const DEFAULT_BASE_URL = "https://developers.openai.com"; + +function parseArgs(argv) { + const args = { + source: process.env.LATEST_MODEL_URL || DEFAULT_URL, + baseUrl: process.env.LATEST_MODEL_BASE_URL || DEFAULT_BASE_URL, + }; + + for (let i = 2; i < argv.length; i += 1) { + const arg = argv[i]; + if (arg === "--source" || arg === "--url") { + args.source = argv[i + 1]; + i += 1; + } else if (arg === "--base-url") { + args.baseUrl = argv[i + 1]; + i += 1; + } + } + + return args; +} + +async function readSource(source) { + if (source.startsWith("file://")) { + return fs.readFile(new URL(source), "utf8"); + } + + if (!/^https?:\/\//.test(source)) { + return fs.readFile(path.resolve(source), "utf8"); + } + + const response = await fetch(source, { + headers: { accept: "text/markdown,text/plain,*/*" }, + }); + + if (!response.ok) { + throw new Error(`failed to fetch ${source}: ${response.status}`); + } + + return response.text(); +} + +function parseIndentedInfo(lines, startIndex) { + const info = {}; + + for (let i = startIndex + 1; i < lines.length; i += 1) { + const line = lines[i]; + if (!line.trim()) { + continue; + } + + const match = line.match(/^ {2}([A-Za-z][A-Za-z0-9_-]*):\s*(.+?)\s*$/); + if (!match) { + break; + } + + info[match[1]] = match[2].replace(/^["']|["']$/g, ""); + } + + return info; +} + +function parseFlatInfo(block) { + const info = {}; + + for (const line of block.split(/\r?\n/)) { + const match = line.match(/^\s*([A-Za-z][A-Za-z0-9_-]*):\s*(.+?)\s*$/); + if (match) { + info[match[1]] = match[2].replace(/^["']|["']$/g, ""); + } + } + + return info; +} + +function extractLatestModelInfo(markdown) { + const lines = markdown.split(/\r?\n/); + const latestModelInfoIndex = lines.findIndex((line) => + /^latestModelInfo:\s*$/.test(line) + ); + + if (latestModelInfoIndex >= 0) { + return parseIndentedInfo(lines, latestModelInfoIndex); + } + + const commentMatch = markdown.match( + //m + ); + if (commentMatch) { + return parseFlatInfo(commentMatch[1]); + } + + return undefined; +} + +function modelToSkillSlug(model) { + return model.trim().replace(/\./g, "p"); +} + +function absoluteUrl(baseUrl, value) { + return new URL(value, baseUrl).toString(); +} + +function normalizeInfo(info, baseUrl) { + const model = info?.model?.trim(); + const migrationGuide = info?.migrationGuide?.trim(); + const promptingGuide = info?.promptingGuide?.trim(); + + if (!model || !migrationGuide || !promptingGuide) { + throw new Error( + "latestModelInfo must include model, migrationGuide, and promptingGuide" + ); + } + + return { + model, + modelSlug: modelToSkillSlug(model), + migrationGuideUrl: absoluteUrl(baseUrl, migrationGuide), + promptingGuideUrl: absoluteUrl(baseUrl, promptingGuide), + }; +} + +async function main() { + const { source, baseUrl } = parseArgs(process.argv); + const markdown = await readSource(source); + const info = extractLatestModelInfo(markdown); + + if (!info) { + throw new Error(`latestModelInfo block not found in ${source}`); + } + + process.stdout.write( + `${JSON.stringify(normalizeInfo(info, baseUrl), null, 2)}\n` + ); +} + +main().catch((error) => { + console.error(error.message); + process.exit(1); +}); diff --git a/.github/skills/openai-docx/SKILL.md b/.github/skills/openai-docx/SKILL.md deleted file mode 100644 index bc730f10..00000000 --- a/.github/skills/openai-docx/SKILL.md +++ /dev/null @@ -1,80 +0,0 @@ ---- -name: openai-docx -description: "Use when the task involves reading, creating, or editing `.docx` documents, especially when formatting or layout fidelity matters; prefer `python-docx` plus the bundled `scripts/render_docx.py` for visual checks." ---- - - -# DOCX Skill - -## When to use -- Read or review DOCX content where layout matters (tables, diagrams, pagination). -- Create or edit DOCX files with professional formatting. -- Validate visual layout before delivery. - -## Workflow -1. Prefer visual review (layout, tables, diagrams). - - If `soffice` and `pdftoppm` are available, convert DOCX -> PDF -> PNGs. - - Or use `scripts/render_docx.py` (requires `pdf2image` and Poppler). - - If these tools are missing, install them or ask the user to review rendered pages locally. -2. Use `python-docx` for edits and structured creation (headings, styles, tables, lists). -3. After each meaningful change, re-render and inspect the pages. -4. If visual review is not possible, extract text with `python-docx` as a fallback and call out layout risk. -5. Keep intermediate outputs organized and clean up after final approval. - -## Temp and output conventions -- Use `tmp/docs/` for intermediate files; delete when done. -- Write final artifacts under `output/doc/` when working in this repo. -- Keep filenames stable and descriptive. - -## Dependencies (install if missing) -Prefer `uv` for dependency management. - -Python packages: -``` -uv pip install python-docx pdf2image -``` -If `uv` is unavailable: -``` -python3 -m pip install python-docx pdf2image -``` -System tools (for rendering): -``` -# macOS (Homebrew) -brew install libreoffice poppler - -# Ubuntu/Debian -sudo apt-get install -y libreoffice poppler-utils -``` - -If installation isn't possible in this environment, tell the user which dependency is missing and how to install it locally. - -## Environment -No required environment variables. - -## Rendering commands -DOCX -> PDF: -``` -soffice -env:UserInstallation=file:///tmp/lo_profile_$$ --headless --convert-to pdf --outdir $OUTDIR $INPUT_DOCX -``` - -PDF -> PNGs: -``` -pdftoppm -png $OUTDIR/$BASENAME.pdf $OUTDIR/$BASENAME -``` - -Bundled helper: -``` -python3 scripts/render_docx.py /path/to/file.docx --output_dir /tmp/docx_pages -``` - -## Quality expectations -- Deliver a client-ready document: consistent typography, spacing, margins, and clear hierarchy. -- Avoid formatting defects: clipped/overlapping text, broken tables, unreadable characters, or default-template styling. -- Charts, tables, and visuals must be legible in rendered pages with correct alignment. -- Use ASCII hyphens only. Avoid U+2011 (non-breaking hyphen) and other Unicode dashes. -- Citations and references must be human-readable; never leave tool tokens or placeholder strings. - -## Final checks -- Re-render and inspect every page at 100% zoom before final delivery. -- Fix any spacing, alignment, or pagination issues and repeat the render loop. -- Confirm there are no leftovers (temp files, duplicate renders) unless the user asks to keep them. diff --git a/.github/skills/openai-docx/agents/openai.yaml b/.github/skills/openai-docx/agents/openai.yaml deleted file mode 100644 index 27ce451b..00000000 --- a/.github/skills/openai-docx/agents/openai.yaml +++ /dev/null @@ -1,6 +0,0 @@ -interface: - display_name: "Word Docs" - short_description: "Edit and review docx files" - icon_small: "./assets/doc-small.svg" - icon_large: "./assets/doc.png" - default_prompt: "Edit or review this .docx file and return the updated file plus a concise change summary." diff --git a/.github/skills/openai-docx/assets/doc-small.svg b/.github/skills/openai-docx/assets/doc-small.svg deleted file mode 100644 index 97289eb2..00000000 --- a/.github/skills/openai-docx/assets/doc-small.svg +++ /dev/null @@ -1,3 +0,0 @@ - - - diff --git a/.github/skills/openai-docx/assets/doc.png b/.github/skills/openai-docx/assets/doc.png deleted file mode 100644 index e1651789..00000000 Binary files a/.github/skills/openai-docx/assets/doc.png and /dev/null differ diff --git a/.github/skills/openai-docx/scripts/render_docx.py b/.github/skills/openai-docx/scripts/render_docx.py deleted file mode 100755 index be4d0d3e..00000000 --- a/.github/skills/openai-docx/scripts/render_docx.py +++ /dev/null @@ -1,297 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import os -import re -import subprocess -import tempfile -import xml.etree.ElementTree as ET -from os import makedirs, replace -from os.path import abspath, basename, exists, expanduser, join, splitext -from shutil import which -import sys -from typing import Sequence, cast -from zipfile import ZipFile - -from pdf2image import convert_from_path, pdfinfo_from_path - -TWIPS_PER_INCH: int = 1440 - - -def ensure_system_tools() -> None: - missing: list[str] = [] - for tool in ("soffice", "pdftoppm"): - if which(tool) is None: - missing.append(tool) - if missing: - tools = ", ".join(missing) - raise RuntimeError( - f"Missing required system tool(s): {tools}. Install LibreOffice and Poppler, then retry." - ) - - -def calc_dpi_via_ooxml_docx(input_path: str, max_w_px: int, max_h_px: int) -> int: - """Calculate DPI from OOXML `word/document.xml` page size (w:pgSz in twips). - - DOCX stores page dimensions in section properties as twips (1/1440 inch). - We read the first encountered section's page size and compute an isotropic DPI - that fits within the target max pixel dimensions. - """ - with ZipFile(input_path, "r") as zf: - xml = zf.read("word/document.xml") - root = ET.fromstring(xml) - ns = {"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"} - - # Common placements: w:body/w:sectPr or w:body/w:p/w:pPr/w:sectPr - sect_pr = root.find(".//w:sectPr", ns) - if sect_pr is None: - raise RuntimeError("Section properties not found in document.xml") - pg_sz = sect_pr.find("w:pgSz", ns) - if pg_sz is None: - raise RuntimeError("Page size not found in section properties") - - # Values are in twips - w_twips_str = pg_sz.get( - "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}w" - ) or pg_sz.get("w") - h_twips_str = pg_sz.get( - "{http://schemas.openxmlformats.org/wordprocessingml/2006/main}h" - ) or pg_sz.get("h") - - if not w_twips_str or not h_twips_str: - raise RuntimeError("Page size attributes missing in pgSz") - - width_in = int(w_twips_str) / TWIPS_PER_INCH - height_in = int(h_twips_str) / TWIPS_PER_INCH - if width_in <= 0 or height_in <= 0: - raise RuntimeError("Invalid page size values in document.xml") - return round(min(max_w_px / width_in, max_h_px / height_in)) - - -def calc_dpi_via_pdf(input_path: str, max_w_px: int, max_h_px: int) -> int: - """Convert input to PDF and compute DPI from its page size.""" - with tempfile.TemporaryDirectory(prefix="soffice_profile_") as user_profile: - with tempfile.TemporaryDirectory(prefix="soffice_convert_") as convert_tmp_dir: - stem = splitext(basename(input_path))[0] - pdf_path = convert_to_pdf(input_path, user_profile, convert_tmp_dir, stem) - if not (pdf_path and exists(pdf_path)): - raise RuntimeError("Failed to convert input to PDF for DPI computation.") - - info = pdfinfo_from_path(pdf_path) - size_val = info.get("Page size") - if not size_val: - for k, v in info.items(): - if isinstance(v, str) and "size" in k.lower() and "pts" in v: - size_val = v - break - if not isinstance(size_val, str): - raise RuntimeError("Failed to read PDF page size for DPI computation.") - - m = re.search(r"(\d+)\s*x\s*(\d+)\s*pts", size_val) - if not m: - raise RuntimeError("Unrecognized PDF page size format.") - width_pts = int(m.group(1)) - height_pts = int(m.group(2)) - width_in = width_pts / 72.0 - height_in = height_pts / 72.0 - if width_in <= 0 or height_in <= 0: - raise RuntimeError("Invalid PDF page size values.") - return round(min(max_w_px / width_in, max_h_px / height_in)) - - -def run_cmd_no_check(cmd: list[str]) -> None: - subprocess.run( - cmd, - check=False, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - env=os.environ.copy(), - ) - - -def convert_to_pdf( - doc_path: str, - user_profile: str, - convert_tmp_dir: str, - stem: str, -) -> str: - # Try direct DOC(X) -> PDF - cmd_pdf = [ - "soffice", - "-env:UserInstallation=file://" + user_profile, - "--invisible", - "--headless", - "--norestore", - "--convert-to", - "pdf", - "--outdir", - convert_tmp_dir, - doc_path, - ] - run_cmd_no_check(cmd_pdf) - - pdf_path = join(convert_tmp_dir, f"{stem}.pdf") - if exists(pdf_path): - return pdf_path - - # Fallback: DOCX -> ODT, then ODT -> PDF - cmd_odt = [ - "soffice", - "-env:UserInstallation=file://" + user_profile, - "--invisible", - "--headless", - "--norestore", - "--convert-to", - "odt", - "--outdir", - convert_tmp_dir, - doc_path, - ] - run_cmd_no_check(cmd_odt) - - odt_path = join(convert_tmp_dir, f"{stem}.odt") - - if exists(odt_path): - cmd_odt_pdf = [ - "soffice", - "-env:UserInstallation=file://" + user_profile, - "--invisible", - "--headless", - "--norestore", - "--convert-to", - "pdf", - "--outdir", - convert_tmp_dir, - odt_path, - ] - run_cmd_no_check(cmd_odt_pdf) - if exists(pdf_path): - return pdf_path - - return "" - - -def rasterize( - doc_path: str, - out_dir: str, - dpi: int, -) -> Sequence[str]: - """Rasterise DOCX (or similar) to images placed in out_dir and return their paths. - - Images are named as page-. with pages starting at 1. - """ - makedirs(out_dir, exist_ok=True) - doc_path = abspath(doc_path) - stem = splitext(basename(doc_path))[0] - - # Use a unique user profile to avoid LibreOffice profile lock when running concurrently - with tempfile.TemporaryDirectory(prefix="soffice_profile_") as user_profile: - # Write conversion outputs into a temp directory to avoid any IO oddities - with tempfile.TemporaryDirectory(prefix="soffice_convert_") as convert_tmp_dir: - pdf_path = convert_to_pdf( - doc_path, - user_profile, - convert_tmp_dir, - stem, - ) - - if not pdf_path or not exists(pdf_path): - raise RuntimeError( - "Failed to produce PDF for rasterization (direct and ODT fallback)." - ) - paths_raw = cast( - list[str], - convert_from_path( - pdf_path, - dpi=dpi, - fmt="png", - thread_count=8, - output_folder=out_dir, - paths_only=True, - output_file="page", - ), - ) - - # Rename convert_from_path's output format f'page{thread_id:04d}-{page_num:02d}.' to 'page-.' - pages: list[tuple[int, str]] = [] - for src_path in paths_raw: - base = splitext(basename(src_path))[0] - page_num_str = base.split("-")[-1] - page_num = int(page_num_str) - dst_path = join(out_dir, f"page-{page_num}.png") - replace(src_path, dst_path) - pages.append((page_num, dst_path)) - pages.sort(key=lambda t: t[0]) - final_paths = [path for _, path in pages] - return final_paths - - -def main() -> None: - parser = argparse.ArgumentParser(description="Render DOCX-like file to PNG images.") - parser.add_argument( - "input_path", - type=str, - help="Path to the input DOCX file (or compatible).", - ) - parser.add_argument( - "--output_dir", - type=str, - default=None, - help=( - "Output directory for the rendered images. " - "Defaults to a folder next to the input named after the input file (without extension)." - ), - ) - parser.add_argument( - "--width", - type=int, - default=1600, - help=( - "Approximate maximum width in pixels after isotropic scaling (default 1600). " - "The actual value may exceed slightly." - ), - ) - parser.add_argument( - "--height", - type=int, - default=2000, - help=( - "Approximate maximum height in pixels after isotropic scaling (default 2000). " - "The actual value may exceed slightly." - ), - ) - parser.add_argument( - "--dpi", - type=int, - default=None, - help=("Override computed DPI. If provided, skips DOCX/PDF-based DPI calculation."), - ) - args = parser.parse_args() - - try: - ensure_system_tools() - - input_path = abspath(expanduser(args.input_path)) - out_dir = ( - abspath(expanduser(args.output_dir)) if args.output_dir else splitext(input_path)[0] - ) - - if args.dpi is not None: - dpi = int(args.dpi) - else: - try: - if input_path.lower().endswith((".docx", ".docm", ".dotx", ".dotm")): - dpi = calc_dpi_via_ooxml_docx(input_path, args.width, args.height) - else: - raise RuntimeError("Skip OOXML DPI; not a DOCX container") - except Exception: - dpi = calc_dpi_via_pdf(input_path, args.width, args.height) - - rasterize(input_path, out_dir, dpi) - print("Pages rendered to " + out_dir) - except RuntimeError as exc: - print(f"Error: {exc}", file=sys.stderr) - raise SystemExit(1) - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-pdf/LICENSE.txt b/.github/skills/openai-pdf/LICENSE.txt deleted file mode 100644 index 13e25df8..00000000 --- a/.github/skills/openai-pdf/LICENSE.txt +++ /dev/null @@ -1,201 +0,0 @@ -Apache License -Version 2.0, January 2004 -http://www.apache.org/licenses/ - -TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - -1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - -2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - -3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - -4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - -5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - -6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - -7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - -8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - -9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf of - any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - -END OF TERMS AND CONDITIONS - -APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don\'t include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - -Copyright [yyyy] [name of copyright owner] - -Licensed under the Apache License, Version 2.0 (the "License"); -you may not use this file except in compliance with the License. -You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - -Unless required by applicable law or agreed to in writing, software -distributed under the License is distributed on an "AS IS" BASIS, -WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -See the License for the specific language governing permissions and -limitations under the License. diff --git a/.github/skills/openai-pdf/SKILL.md b/.github/skills/openai-pdf/SKILL.md deleted file mode 100644 index f1ca0812..00000000 --- a/.github/skills/openai-pdf/SKILL.md +++ /dev/null @@ -1,67 +0,0 @@ ---- -name: openai-pdf -description: "Use when tasks involve reading, creating, or reviewing PDF files where rendering and layout matter; prefer visual checks by rendering pages (Poppler) and use Python tools such as `reportlab`, `pdfplumber`, and `pypdf` for generation and extraction." ---- - - -# PDF Skill - -## When to use -- Read or review PDF content where layout and visuals matter. -- Create PDFs programmatically with reliable formatting. -- Validate final rendering before delivery. - -## Workflow -1. Prefer visual review: render PDF pages to PNGs and inspect them. - - Use `pdftoppm` if available. - - If unavailable, install Poppler or ask the user to review the output locally. -2. Use `reportlab` to generate PDFs when creating new documents. -3. Use `pdfplumber` (or `pypdf`) for text extraction and quick checks; do not rely on it for layout fidelity. -4. After each meaningful update, re-render pages and verify alignment, spacing, and legibility. - -## Temp and output conventions -- Use `tmp/pdfs/` for intermediate files; delete when done. -- Write final artifacts under `output/pdf/` when working in this repo. -- Keep filenames stable and descriptive. - -## Dependencies (install if missing) -Prefer `uv` for dependency management. - -Python packages: -``` -uv pip install reportlab pdfplumber pypdf -``` -If `uv` is unavailable: -``` -python3 -m pip install reportlab pdfplumber pypdf -``` -System tools (for rendering): -``` -# macOS (Homebrew) -brew install poppler - -# Ubuntu/Debian -sudo apt-get install -y poppler-utils -``` - -If installation isn't possible in this environment, tell the user which dependency is missing and how to install it locally. - -## Environment -No required environment variables. - -## Rendering command -``` -pdftoppm -png $INPUT_PDF $OUTPUT_PREFIX -``` - -## Quality expectations -- Maintain polished visual design: consistent typography, spacing, margins, and section hierarchy. -- Avoid rendering issues: clipped text, overlapping elements, broken tables, black squares, or unreadable glyphs. -- Charts, tables, and images must be sharp, aligned, and clearly labeled. -- Use ASCII hyphens only. Avoid U+2011 (non-breaking hyphen) and other Unicode dashes. -- Citations and references must be human-readable; never leave tool tokens or placeholder strings. - -## Final checks -- Do not deliver until the latest PNG inspection shows zero visual or formatting defects. -- Confirm headers/footers, page numbering, and section transitions look polished. -- Keep intermediate files organized or remove them after final approval. diff --git a/.github/skills/openai-pdf/agents/openai.yaml b/.github/skills/openai-pdf/agents/openai.yaml deleted file mode 100644 index fe2876a6..00000000 --- a/.github/skills/openai-pdf/agents/openai.yaml +++ /dev/null @@ -1,5 +0,0 @@ -interface: - display_name: "PDF Skill" - short_description: "Create, edit, and review PDFs" - icon_large: "./assets/pdf.png" - default_prompt: "Create, edit, or review this PDF and summarize the key output or changes." diff --git a/.github/skills/openai-pdf/assets/pdf.png b/.github/skills/openai-pdf/assets/pdf.png deleted file mode 100644 index dd16ba28..00000000 Binary files a/.github/skills/openai-pdf/assets/pdf.png and /dev/null differ diff --git a/.github/skills/openai-skill-creator/LICENSE.txt b/.github/skills/openai-skill-creator/LICENSE.txt deleted file mode 100644 index d6456956..00000000 --- a/.github/skills/openai-skill-creator/LICENSE.txt +++ /dev/null @@ -1,202 +0,0 @@ - - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - - Copyright [yyyy] [name of copyright owner] - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. diff --git a/.github/skills/openai-skill-creator/SKILL.md b/.github/skills/openai-skill-creator/SKILL.md deleted file mode 100644 index a2beb247..00000000 --- a/.github/skills/openai-skill-creator/SKILL.md +++ /dev/null @@ -1,368 +0,0 @@ ---- -name: openai-skill-creator -description: Guide for creating effective skills. This skill should be used when users want to create a new skill (or update an existing skill) that extends Codex's capabilities with specialized knowledge, workflows, or tool integrations. -metadata: - short-description: Create or update a skill ---- - -# Skill Creator - -This skill provides guidance for creating effective skills. - -## About Skills - -Skills are modular, self-contained folders that extend Codex's capabilities by providing -specialized knowledge, workflows, and tools. Think of them as "onboarding guides" for specific -domains or tasks—they transform Codex from a general-purpose agent into a specialized agent -equipped with procedural knowledge that no model can fully possess. - -### What Skills Provide - -1. Specialized workflows - Multi-step procedures for specific domains -2. Tool integrations - Instructions for working with specific file formats or APIs -3. Domain expertise - Company-specific knowledge, schemas, business logic -4. Bundled resources - Scripts, references, and assets for complex and repetitive tasks - -## Core Principles - -### Concise is Key - -The context window is a public good. Skills share the context window with everything else Codex needs: system prompt, conversation history, other Skills' metadata, and the actual user request. - -**Default assumption: Codex is already very smart.** Only add context Codex doesn't already have. Challenge each piece of information: "Does Codex really need this explanation?" and "Does this paragraph justify its token cost?" - -Prefer concise examples over verbose explanations. - -### Set Appropriate Degrees of Freedom - -Match the level of specificity to the task's fragility and variability: - -**High freedom (text-based instructions)**: Use when multiple approaches are valid, decisions depend on context, or heuristics guide the approach. - -**Medium freedom (pseudocode or scripts with parameters)**: Use when a preferred pattern exists, some variation is acceptable, or configuration affects behavior. - -**Low freedom (specific scripts, few parameters)**: Use when operations are fragile and error-prone, consistency is critical, or a specific sequence must be followed. - -Think of Codex as exploring a path: a narrow bridge with cliffs needs specific guardrails (low freedom), while an open field allows many routes (high freedom). - -### Anatomy of a Skill - -Every skill consists of a required SKILL.md file and optional bundled resources: - -``` -skill-name/ -├── SKILL.md (required) -│ ├── YAML frontmatter metadata (required) -│ │ ├── name: (required) -│ │ └── description: (required) -│ └── Markdown instructions (required) -├── agents/ (recommended) -│ └── openai.yaml - UI metadata for skill lists and chips -└── Bundled Resources (optional) - ├── scripts/ - Executable code (Python/Bash/etc.) - ├── references/ - Documentation intended to be loaded into context as needed - └── assets/ - Files used in output (templates, icons, fonts, etc.) -``` - -#### SKILL.md (required) - -Every SKILL.md consists of: - -- **Frontmatter** (YAML): Contains `name` and `description` fields. These are the only fields that Codex reads to determine when the skill gets used, thus it is very important to be clear and comprehensive in describing what the skill is, and when it should be used. -- **Body** (Markdown): Instructions and guidance for using the skill. Only loaded AFTER the skill triggers (if at all). - -#### Agents metadata (recommended) - -- UI-facing metadata for skill lists and chips -- Read references/openai_yaml.md before generating values and follow its descriptions and constraints -- Create: human-facing `display_name`, `short_description`, and `default_prompt` by reading the skill -- Generate deterministically by passing the values as `--interface key=value` to `scripts/generate_openai_yaml.py` or `scripts/init_skill.py` -- On updates: validate `agents/openai.yaml` still matches SKILL.md; regenerate if stale -- Only include other optional interface fields (icons, brand color) if explicitly provided -- See references/openai_yaml.md for field definitions and examples - -#### Bundled Resources (optional) - -##### Scripts (`scripts/`) - -Executable code (Python/Bash/etc.) for tasks that require deterministic reliability or are repeatedly rewritten. - -- **When to include**: When the same code is being rewritten repeatedly or deterministic reliability is needed -- **Example**: `scripts/rotate_pdf.py` for PDF rotation tasks -- **Benefits**: Token efficient, deterministic, may be executed without loading into context -- **Note**: Scripts may still need to be read by Codex for patching or environment-specific adjustments - -##### References (`references/`) - -Documentation and reference material intended to be loaded as needed into context to inform Codex's process and thinking. - -- **When to include**: For documentation that Codex should reference while working -- **Examples**: `references/finance.md` for financial schemas, `references/mnda.md` for company NDA template, `references/policies.md` for company policies, `references/api_docs.md` for API specifications -- **Use cases**: Database schemas, API documentation, domain knowledge, company policies, detailed workflow guides -- **Benefits**: Keeps SKILL.md lean, loaded only when Codex determines it's needed -- **Best practice**: If files are large (>10k words), include grep search patterns in SKILL.md -- **Avoid duplication**: Information should live in either SKILL.md or references files, not both. Prefer references files for detailed information unless it's truly core to the skill—this keeps SKILL.md lean while making information discoverable without hogging the context window. Keep only essential procedural instructions and workflow guidance in SKILL.md; move detailed reference material, schemas, and examples to references files. - -##### Assets (`assets/`) - -Files not intended to be loaded into context, but rather used within the output Codex produces. - -- **When to include**: When the skill needs files that will be used in the final output -- **Examples**: `assets/logo.png` for brand assets, `assets/slides.pptx` for PowerPoint templates, `assets/frontend-template/` for HTML/React boilerplate, `assets/font.ttf` for typography -- **Use cases**: Templates, images, icons, boilerplate code, fonts, sample documents that get copied or modified -- **Benefits**: Separates output resources from documentation, enables Codex to use files without loading them into context - -#### What to Not Include in a Skill - -A skill should only contain essential files that directly support its functionality. Do NOT create extraneous documentation or auxiliary files, including: - -- README.md -- INSTALLATION_GUIDE.md -- QUICK_REFERENCE.md -- CHANGELOG.md -- etc. - -The skill should only contain the information needed for an AI agent to do the job at hand. It should not contain auxiliary context about the process that went into creating it, setup and testing procedures, user-facing documentation, etc. Creating additional documentation files just adds clutter and confusion. - -### Progressive Disclosure Design Principle - -Skills use a three-level loading system to manage context efficiently: - -1. **Metadata (name + description)** - Always in context (~100 words) -2. **SKILL.md body** - When skill triggers (<5k words) -3. **Bundled resources** - As needed by Codex (Unlimited because scripts can be executed without reading into context window) - -#### Progressive Disclosure Patterns - -Keep SKILL.md body to the essentials and under 500 lines to minimize context bloat. Split content into separate files when approaching this limit. When splitting out content into other files, it is very important to reference them from SKILL.md and describe clearly when to read them, to ensure the reader of the skill knows they exist and when to use them. - -**Key principle:** When a skill supports multiple variations, frameworks, or options, keep only the core workflow and selection guidance in SKILL.md. Move variant-specific details (patterns, examples, configuration) into separate reference files. - -**Pattern 1: High-level guide with references** - -```markdown -# PDF Processing - -## Quick start - -Extract text with pdfplumber: -[code example] - -## Advanced features - -- **Form filling**: See [FORMS.md](FORMS.md) for complete guide -- **API reference**: See [REFERENCE.md](REFERENCE.md) for all methods -- **Examples**: See [EXAMPLES.md](EXAMPLES.md) for common patterns -``` - -Codex loads FORMS.md, REFERENCE.md, or EXAMPLES.md only when needed. - -**Pattern 2: Domain-specific organization** - -For Skills with multiple domains, organize content by domain to avoid loading irrelevant context: - -``` -bigquery-skill/ -├── SKILL.md (overview and navigation) -└── reference/ - ├── finance.md (revenue, billing metrics) - ├── sales.md (opportunities, pipeline) - ├── product.md (API usage, features) - └── marketing.md (campaigns, attribution) -``` - -When a user asks about sales metrics, Codex only reads sales.md. - -Similarly, for skills supporting multiple frameworks or variants, organize by variant: - -``` -cloud-deploy/ -├── SKILL.md (workflow + provider selection) -└── references/ - ├── aws.md (AWS deployment patterns) - ├── gcp.md (GCP deployment patterns) - └── azure.md (Azure deployment patterns) -``` - -When the user chooses AWS, Codex only reads aws.md. - -**Pattern 3: Conditional details** - -Show basic content, link to advanced content: - -```markdown -# DOCX Processing - -## Creating documents - -Use docx-js for new documents. See [DOCX-JS.md](DOCX-JS.md). - -## Editing documents - -For simple edits, modify the XML directly. - -**For tracked changes**: See [REDLINING.md](REDLINING.md) -**For OOXML details**: See [OOXML.md](OOXML.md) -``` - -Codex reads REDLINING.md or OOXML.md only when the user needs those features. - -**Important guidelines:** - -- **Avoid deeply nested references** - Keep references one level deep from SKILL.md. All reference files should link directly from SKILL.md. -- **Structure longer reference files** - For files longer than 100 lines, include a table of contents at the top so Codex can see the full scope when previewing. - -## Skill Creation Process - -Skill creation involves these steps: - -1. Understand the skill with concrete examples -2. Plan reusable skill contents (scripts, references, assets) -3. Initialize the skill (run init_skill.py) -4. Edit the skill (implement resources and write SKILL.md) -5. Validate the skill (run quick_validate.py) -6. Iterate based on real usage - -Follow these steps in order, skipping only if there is a clear reason why they are not applicable. - -### Skill Naming - -- Use lowercase letters, digits, and hyphens only; normalize user-provided titles to hyphen-case (e.g., "Plan Mode" -> `plan-mode`). -- When generating names, generate a name under 64 characters (letters, digits, hyphens). -- Prefer short, verb-led phrases that describe the action. -- Namespace by tool when it improves clarity or triggering (e.g., `gh-address-comments`, `linear-address-issue`). -- Name the skill folder exactly after the skill name. - -### Step 1: Understanding the Skill with Concrete Examples - -Skip this step only when the skill's usage patterns are already clearly understood. It remains valuable even when working with an existing skill. - -To create an effective skill, clearly understand concrete examples of how the skill will be used. This understanding can come from either direct user examples or generated examples that are validated with user feedback. - -For example, when building an image-editor skill, relevant questions include: - -- "What functionality should the image-editor skill support? Editing, rotating, anything else?" -- "Can you give some examples of how this skill would be used?" -- "I can imagine users asking for things like 'Remove the red-eye from this image' or 'Rotate this image'. Are there other ways you imagine this skill being used?" -- "What would a user say that should trigger this skill?" - -To avoid overwhelming users, avoid asking too many questions in a single message. Start with the most important questions and follow up as needed for better effectiveness. - -Conclude this step when there is a clear sense of the functionality the skill should support. - -### Step 2: Planning the Reusable Skill Contents - -To turn concrete examples into an effective skill, analyze each example by: - -1. Considering how to execute on the example from scratch -2. Identifying what scripts, references, and assets would be helpful when executing these workflows repeatedly - -Example: When building a `pdf-editor` skill to handle queries like "Help me rotate this PDF," the analysis shows: - -1. Rotating a PDF requires re-writing the same code each time -2. A `scripts/rotate_pdf.py` script would be helpful to store in the skill - -Example: When designing a `frontend-webapp-builder` skill for queries like "Build me a todo app" or "Build me a dashboard to track my steps," the analysis shows: - -1. Writing a frontend webapp requires the same boilerplate HTML/React each time -2. An `assets/hello-world/` template containing the boilerplate HTML/React project files would be helpful to store in the skill - -Example: When building a `big-query` skill to handle queries like "How many users have logged in today?" the analysis shows: - -1. Querying BigQuery requires re-discovering the table schemas and relationships each time -2. A `references/schema.md` file documenting the table schemas would be helpful to store in the skill - -To establish the skill's contents, analyze each concrete example to create a list of the reusable resources to include: scripts, references, and assets. - -### Step 3: Initializing the Skill - -At this point, it is time to actually create the skill. - -Skip this step only if the skill being developed already exists. In this case, continue to the next step. - -When creating a new skill from scratch, always run the `init_skill.py` script. The script conveniently generates a new template skill directory that automatically includes everything a skill requires, making the skill creation process much more efficient and reliable. - -Usage: - -```bash -scripts/init_skill.py --path [--resources scripts,references,assets] [--examples] -``` - -Examples: - -```bash -scripts/init_skill.py my-skill --path skills/public -scripts/init_skill.py my-skill --path skills/public --resources scripts,references -scripts/init_skill.py my-skill --path skills/public --resources scripts --examples -``` - -The script: - -- Creates the skill directory at the specified path -- Generates a SKILL.md template with proper frontmatter and TODO placeholders -- Creates `agents/openai.yaml` using agent-generated `display_name`, `short_description`, and `default_prompt` passed via `--interface key=value` -- Optionally creates resource directories based on `--resources` -- Optionally adds example files when `--examples` is set - -After initialization, customize the SKILL.md and add resources as needed. If you used `--examples`, replace or delete placeholder files. - -Generate `display_name`, `short_description`, and `default_prompt` by reading the skill, then pass them as `--interface key=value` to `init_skill.py` or regenerate with: - -```bash -scripts/generate_openai_yaml.py --interface key=value -``` - -Only include other optional interface fields when the user explicitly provides them. For full field descriptions and examples, see references/openai_yaml.md. - -### Step 4: Edit the Skill - -When editing the (newly-generated or existing) skill, remember that the skill is being created for another instance of Codex to use. Include information that would be beneficial and non-obvious to Codex. Consider what procedural knowledge, domain-specific details, or reusable assets would help another Codex instance execute these tasks more effectively. - -#### Start with Reusable Skill Contents - -To begin implementation, start with the reusable resources identified above: `scripts/`, `references/`, and `assets/` files. Note that this step may require user input. For example, when implementing a `brand-guidelines` skill, the user may need to provide brand assets or templates to store in `assets/`, or documentation to store in `references/`. - -Added scripts must be tested by actually running them to ensure there are no bugs and that the output matches what is expected. If there are many similar scripts, only a representative sample needs to be tested to ensure confidence that they all work while balancing time to completion. - -If you used `--examples`, delete any placeholder files that are not needed for the skill. Only create resource directories that are actually required. - -#### Update SKILL.md - -**Writing Guidelines:** Always use imperative/infinitive form. - -##### Frontmatter - -Write the YAML frontmatter with `name` and `description`: - -- `name`: The skill name -- `description`: This is the primary triggering mechanism for your skill, and helps Codex understand when to use the skill. - - Include both what the Skill does and specific triggers/contexts for when to use it. - - Include all "when to use" information here - Not in the body. The body is only loaded after triggering, so "When to Use This Skill" sections in the body are not helpful to Codex. - - Example description for a `docx` skill: "Comprehensive document creation, editing, and analysis with support for tracked changes, comments, formatting preservation, and text extraction. Use when Codex needs to work with professional documents (.docx files) for: (1) Creating new documents, (2) Modifying or editing content, (3) Working with tracked changes, (4) Adding comments, or any other document tasks" - -Do not include any other fields in YAML frontmatter. - -##### Body - -Write instructions for using the skill and its bundled resources. - -### Step 5: Validate the Skill - -Once development of the skill is complete, validate the skill folder to catch basic issues early: - -```bash -scripts/quick_validate.py -``` - -The validation script checks YAML frontmatter format, required fields, and naming rules. If validation fails, fix the reported issues and run the command again. - -### Step 6: Iterate - -After testing the skill, users may request improvements. Often this happens right after using the skill, with fresh context of how the skill performed. - -**Iteration workflow:** - -1. Use the skill on real tasks -2. Notice struggles or inefficiencies -3. Identify how SKILL.md or bundled resources should be updated -4. Implement changes and test again diff --git a/.github/skills/openai-skill-creator/agents/openai.yaml b/.github/skills/openai-skill-creator/agents/openai.yaml deleted file mode 100644 index f5ce09ed..00000000 --- a/.github/skills/openai-skill-creator/agents/openai.yaml +++ /dev/null @@ -1,4 +0,0 @@ -interface: - display_name: "Skill Creator" - short_description: "Create or update a skill" - default_prompt: "Read my repository and create a skill to bootstrap new components for my project." \ No newline at end of file diff --git a/.github/skills/openai-skill-creator/references/openai_yaml.md b/.github/skills/openai-skill-creator/references/openai_yaml.md deleted file mode 100644 index da5629f8..00000000 --- a/.github/skills/openai-skill-creator/references/openai_yaml.md +++ /dev/null @@ -1,43 +0,0 @@ -# openai.yaml fields (full example + descriptions) - -`agents/openai.yaml` is an extended, product-specific config intended for the machine/harness to read, not the agent. Other product-specific config can also live in the `agents/` folder. - -## Full example - -```yaml -interface: - display_name: "Optional user-facing name" - short_description: "Optional user-facing description" - icon_small: "./assets/small-400px.png" - icon_large: "./assets/large-logo.svg" - brand_color: "#3B82F6" - default_prompt: "Optional surrounding prompt to use the skill with" - -dependencies: - tools: - - type: "mcp" - value: "github" - description: "GitHub MCP server" - transport: "streamable_http" - url: "https://api.githubcopilot.com/mcp/" -``` - -## Field descriptions and constraints - -Top-level constraints: - -- Quote all string values. -- Keep keys unquoted. -- For `interface.default_prompt`: generate a helpful, short (typically 1 sentence) example starting prompt based on the skill. It must explicitly mention the skill as `$skill-name` (e.g., "Use $skill-name-here to draft a concise weekly status update."). - -- `interface.display_name`: Human-facing title shown in UI skill lists and chips. -- `interface.short_description`: Human-facing short UI blurb (25–64 chars) for quick scanning. -- `interface.icon_small`: Path to a small icon asset (relative to skill dir). Default to `./assets/` and place icons in the skill's `assets/` folder. -- `interface.icon_large`: Path to a larger logo asset (relative to skill dir). Default to `./assets/` and place icons in the skill's `assets/` folder. -- `interface.brand_color`: Hex color used for UI accents (e.g., badges). -- `interface.default_prompt`: Default prompt snippet inserted when invoking the skill. -- `dependencies.tools[].type`: Dependency category. Only `mcp` is supported for now. -- `dependencies.tools[].value`: Identifier of the tool or dependency. -- `dependencies.tools[].description`: Human-readable explanation of the dependency. -- `dependencies.tools[].transport`: Connection type when `type` is `mcp`. -- `dependencies.tools[].url`: MCP server URL when `type` is `mcp`. diff --git a/.github/skills/openai-skill-creator/scripts/generate_openai_yaml.py b/.github/skills/openai-skill-creator/scripts/generate_openai_yaml.py deleted file mode 100644 index 1a9d784f..00000000 --- a/.github/skills/openai-skill-creator/scripts/generate_openai_yaml.py +++ /dev/null @@ -1,225 +0,0 @@ -#!/usr/bin/env python3 -""" -OpenAI YAML Generator - Creates agents/openai.yaml for a skill folder. - -Usage: - generate_openai_yaml.py [--name ] [--interface key=value] -""" - -import argparse -import re -import sys -from pathlib import Path - -import yaml - -ACRONYMS = { - "GH", - "MCP", - "API", - "CI", - "CLI", - "LLM", - "PDF", - "PR", - "UI", - "URL", - "SQL", -} - -BRANDS = { - "openai": "OpenAI", - "openapi": "OpenAPI", - "github": "GitHub", - "pagerduty": "PagerDuty", - "datadog": "DataDog", - "sqlite": "SQLite", - "fastapi": "FastAPI", -} - -SMALL_WORDS = {"and", "or", "to", "up", "with"} - -ALLOWED_INTERFACE_KEYS = { - "display_name", - "short_description", - "icon_small", - "icon_large", - "brand_color", - "default_prompt", -} - - -def yaml_quote(value): - escaped = value.replace("\\", "\\\\").replace('"', '\\"').replace("\n", "\\n") - return f'"{escaped}"' - - -def format_display_name(skill_name): - words = [word for word in skill_name.split("-") if word] - formatted = [] - for index, word in enumerate(words): - lower = word.lower() - upper = word.upper() - if upper in ACRONYMS: - formatted.append(upper) - continue - if lower in BRANDS: - formatted.append(BRANDS[lower]) - continue - if index > 0 and lower in SMALL_WORDS: - formatted.append(lower) - continue - formatted.append(word.capitalize()) - return " ".join(formatted) - - -def generate_short_description(display_name): - description = f"Help with {display_name} tasks" - - if len(description) < 25: - description = f"Help with {display_name} tasks and workflows" - if len(description) < 25: - description = f"Help with {display_name} tasks with guidance" - - if len(description) > 64: - description = f"Help with {display_name}" - if len(description) > 64: - description = f"{display_name} helper" - if len(description) > 64: - description = f"{display_name} tools" - if len(description) > 64: - suffix = " helper" - max_name_length = 64 - len(suffix) - trimmed = display_name[:max_name_length].rstrip() - description = f"{trimmed}{suffix}" - if len(description) > 64: - description = description[:64].rstrip() - - if len(description) < 25: - description = f"{description} workflows" - if len(description) > 64: - description = description[:64].rstrip() - - return description - - -def read_frontmatter_name(skill_dir): - skill_md = Path(skill_dir) / "SKILL.md" - if not skill_md.exists(): - print(f"[ERROR] SKILL.md not found in {skill_dir}") - return None - content = skill_md.read_text() - match = re.match(r"^---\n(.*?)\n---", content, re.DOTALL) - if not match: - print("[ERROR] Invalid SKILL.md frontmatter format.") - return None - frontmatter_text = match.group(1) - try: - frontmatter = yaml.safe_load(frontmatter_text) - except yaml.YAMLError as exc: - print(f"[ERROR] Invalid YAML frontmatter: {exc}") - return None - if not isinstance(frontmatter, dict): - print("[ERROR] Frontmatter must be a YAML dictionary.") - return None - name = frontmatter.get("name", "") - if not isinstance(name, str) or not name.strip(): - print("[ERROR] Frontmatter 'name' is missing or invalid.") - return None - return name.strip() - - -def parse_interface_overrides(raw_overrides): - overrides = {} - optional_order = [] - for item in raw_overrides: - if "=" not in item: - print(f"[ERROR] Invalid interface override '{item}'. Use key=value.") - return None, None - key, value = item.split("=", 1) - key = key.strip() - value = value.strip() - if not key: - print(f"[ERROR] Invalid interface override '{item}'. Key is empty.") - return None, None - if key not in ALLOWED_INTERFACE_KEYS: - allowed = ", ".join(sorted(ALLOWED_INTERFACE_KEYS)) - print(f"[ERROR] Unknown interface field '{key}'. Allowed: {allowed}") - return None, None - overrides[key] = value - if key not in ("display_name", "short_description") and key not in optional_order: - optional_order.append(key) - return overrides, optional_order - - -def write_openai_yaml(skill_dir, skill_name, raw_overrides): - overrides, optional_order = parse_interface_overrides(raw_overrides) - if overrides is None: - return None - - display_name = overrides.get("display_name") or format_display_name(skill_name) - short_description = overrides.get("short_description") or generate_short_description(display_name) - - if not (25 <= len(short_description) <= 64): - print( - "[ERROR] short_description must be 25-64 characters " - f"(got {len(short_description)})." - ) - return None - - interface_lines = [ - "interface:", - f" display_name: {yaml_quote(display_name)}", - f" short_description: {yaml_quote(short_description)}", - ] - - for key in optional_order: - value = overrides.get(key) - if value is not None: - interface_lines.append(f" {key}: {yaml_quote(value)}") - - agents_dir = Path(skill_dir) / "agents" - agents_dir.mkdir(parents=True, exist_ok=True) - output_path = agents_dir / "openai.yaml" - output_path.write_text("\n".join(interface_lines) + "\n") - print(f"[OK] Created agents/openai.yaml") - return output_path - - -def main(): - parser = argparse.ArgumentParser( - description="Create agents/openai.yaml for a skill directory.", - ) - parser.add_argument("skill_dir", help="Path to the skill directory") - parser.add_argument( - "--name", - help="Skill name override (defaults to SKILL.md frontmatter)", - ) - parser.add_argument( - "--interface", - action="append", - default=[], - help="Interface override in key=value format (repeatable)", - ) - args = parser.parse_args() - - skill_dir = Path(args.skill_dir).resolve() - if not skill_dir.exists(): - print(f"[ERROR] Skill directory not found: {skill_dir}") - sys.exit(1) - if not skill_dir.is_dir(): - print(f"[ERROR] Path is not a directory: {skill_dir}") - sys.exit(1) - - skill_name = args.name or read_frontmatter_name(skill_dir) - if not skill_name: - sys.exit(1) - - result = write_openai_yaml(skill_dir, skill_name, args.interface) - if result: - sys.exit(0) - sys.exit(1) - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-skill-creator/scripts/init_skill.py b/.github/skills/openai-skill-creator/scripts/init_skill.py deleted file mode 100644 index f90703ec..00000000 --- a/.github/skills/openai-skill-creator/scripts/init_skill.py +++ /dev/null @@ -1,397 +0,0 @@ -#!/usr/bin/env python3 -""" -Skill Initializer - Creates a new skill from template - -Usage: - init_skill.py --path [--resources scripts,references,assets] [--examples] [--interface key=value] - -Examples: - init_skill.py my-new-skill --path skills/public - init_skill.py my-new-skill --path skills/public --resources scripts,references - init_skill.py my-api-helper --path skills/private --resources scripts --examples - init_skill.py custom-skill --path /custom/location - init_skill.py my-skill --path skills/public --interface short_description="Short UI label" -""" - -import argparse -import re -import sys -from pathlib import Path - -from generate_openai_yaml import write_openai_yaml - -MAX_SKILL_NAME_LENGTH = 64 -ALLOWED_RESOURCES = {"scripts", "references", "assets"} - -SKILL_TEMPLATE = """--- -name: {skill_name} -description: [TODO: Complete and informative explanation of what the skill does and when to use it. Include WHEN to use this skill - specific scenarios, file types, or tasks that trigger it.] ---- - -# {skill_title} - -## Overview - -[TODO: 1-2 sentences explaining what this skill enables] - -## Structuring This Skill - -[TODO: Choose the structure that best fits this skill's purpose. Common patterns: - -**1. Workflow-Based** (best for sequential processes) -- Works well when there are clear step-by-step procedures -- Example: DOCX skill with "Workflow Decision Tree" -> "Reading" -> "Creating" -> "Editing" -- Structure: ## Overview -> ## Workflow Decision Tree -> ## Step 1 -> ## Step 2... - -**2. Task-Based** (best for tool collections) -- Works well when the skill offers different operations/capabilities -- Example: PDF skill with "Quick Start" -> "Merge PDFs" -> "Split PDFs" -> "Extract Text" -- Structure: ## Overview -> ## Quick Start -> ## Task Category 1 -> ## Task Category 2... - -**3. Reference/Guidelines** (best for standards or specifications) -- Works well for brand guidelines, coding standards, or requirements -- Example: Brand styling with "Brand Guidelines" -> "Colors" -> "Typography" -> "Features" -- Structure: ## Overview -> ## Guidelines -> ## Specifications -> ## Usage... - -**4. Capabilities-Based** (best for integrated systems) -- Works well when the skill provides multiple interrelated features -- Example: Product Management with "Core Capabilities" -> numbered capability list -- Structure: ## Overview -> ## Core Capabilities -> ### 1. Feature -> ### 2. Feature... - -Patterns can be mixed and matched as needed. Most skills combine patterns (e.g., start with task-based, add workflow for complex operations). - -Delete this entire "Structuring This Skill" section when done - it's just guidance.] - -## [TODO: Replace with the first main section based on chosen structure] - -[TODO: Add content here. See examples in existing skills: -- Code samples for technical skills -- Decision trees for complex workflows -- Concrete examples with realistic user requests -- References to scripts/templates/references as needed] - -## Resources (optional) - -Create only the resource directories this skill actually needs. Delete this section if no resources are required. - -### scripts/ -Executable code (Python/Bash/etc.) that can be run directly to perform specific operations. - -**Examples from other skills:** -- PDF skill: `fill_fillable_fields.py`, `extract_form_field_info.py` - utilities for PDF manipulation -- DOCX skill: `document.py`, `utilities.py` - Python modules for document processing - -**Appropriate for:** Python scripts, shell scripts, or any executable code that performs automation, data processing, or specific operations. - -**Note:** Scripts may be executed without loading into context, but can still be read by Codex for patching or environment adjustments. - -### references/ -Documentation and reference material intended to be loaded into context to inform Codex's process and thinking. - -**Examples from other skills:** -- Product management: `communication.md`, `context_building.md` - detailed workflow guides -- BigQuery: API reference documentation and query examples -- Finance: Schema documentation, company policies - -**Appropriate for:** In-depth documentation, API references, database schemas, comprehensive guides, or any detailed information that Codex should reference while working. - -### assets/ -Files not intended to be loaded into context, but rather used within the output Codex produces. - -**Examples from other skills:** -- Brand styling: PowerPoint template files (.pptx), logo files -- Frontend builder: HTML/React boilerplate project directories -- Typography: Font files (.ttf, .woff2) - -**Appropriate for:** Templates, boilerplate code, document templates, images, icons, fonts, or any files meant to be copied or used in the final output. - ---- - -**Not every skill requires all three types of resources.** -""" - -EXAMPLE_SCRIPT = '''#!/usr/bin/env python3 -""" -Example helper script for {skill_name} - -This is a placeholder script that can be executed directly. -Replace with actual implementation or delete if not needed. - -Example real scripts from other skills: -- pdf/scripts/fill_fillable_fields.py - Fills PDF form fields -- pdf/scripts/convert_pdf_to_images.py - Converts PDF pages to images -""" - -def main(): - print("This is an example script for {skill_name}") - # TODO: Add actual script logic here - # This could be data processing, file conversion, API calls, etc. - -if __name__ == "__main__": - main() -''' - -EXAMPLE_REFERENCE = """# Reference Documentation for {skill_title} - -This is a placeholder for detailed reference documentation. -Replace with actual reference content or delete if not needed. - -Example real reference docs from other skills: -- product-management/references/communication.md - Comprehensive guide for status updates -- product-management/references/context_building.md - Deep-dive on gathering context -- bigquery/references/ - API references and query examples - -## When Reference Docs Are Useful - -Reference docs are ideal for: -- Comprehensive API documentation -- Detailed workflow guides -- Complex multi-step processes -- Information too lengthy for main SKILL.md -- Content that's only needed for specific use cases - -## Structure Suggestions - -### API Reference Example -- Overview -- Authentication -- Endpoints with examples -- Error codes -- Rate limits - -### Workflow Guide Example -- Prerequisites -- Step-by-step instructions -- Common patterns -- Troubleshooting -- Best practices -""" - -EXAMPLE_ASSET = """# Example Asset File - -This placeholder represents where asset files would be stored. -Replace with actual asset files (templates, images, fonts, etc.) or delete if not needed. - -Asset files are NOT intended to be loaded into context, but rather used within -the output Codex produces. - -Example asset files from other skills: -- Brand guidelines: logo.png, slides_template.pptx -- Frontend builder: hello-world/ directory with HTML/React boilerplate -- Typography: custom-font.ttf, font-family.woff2 -- Data: sample_data.csv, test_dataset.json - -## Common Asset Types - -- Templates: .pptx, .docx, boilerplate directories -- Images: .png, .jpg, .svg, .gif -- Fonts: .ttf, .otf, .woff, .woff2 -- Boilerplate code: Project directories, starter files -- Icons: .ico, .svg -- Data files: .csv, .json, .xml, .yaml - -Note: This is a text placeholder. Actual assets can be any file type. -""" - - -def normalize_skill_name(skill_name): - """Normalize a skill name to lowercase hyphen-case.""" - normalized = skill_name.strip().lower() - normalized = re.sub(r"[^a-z0-9]+", "-", normalized) - normalized = normalized.strip("-") - normalized = re.sub(r"-{2,}", "-", normalized) - return normalized - - -def title_case_skill_name(skill_name): - """Convert hyphenated skill name to Title Case for display.""" - return " ".join(word.capitalize() for word in skill_name.split("-")) - - -def parse_resources(raw_resources): - if not raw_resources: - return [] - resources = [item.strip() for item in raw_resources.split(",") if item.strip()] - invalid = sorted({item for item in resources if item not in ALLOWED_RESOURCES}) - if invalid: - allowed = ", ".join(sorted(ALLOWED_RESOURCES)) - print(f"[ERROR] Unknown resource type(s): {', '.join(invalid)}") - print(f" Allowed: {allowed}") - sys.exit(1) - deduped = [] - seen = set() - for resource in resources: - if resource not in seen: - deduped.append(resource) - seen.add(resource) - return deduped - - -def create_resource_dirs(skill_dir, skill_name, skill_title, resources, include_examples): - for resource in resources: - resource_dir = skill_dir / resource - resource_dir.mkdir(exist_ok=True) - if resource == "scripts": - if include_examples: - example_script = resource_dir / "example.py" - example_script.write_text(EXAMPLE_SCRIPT.format(skill_name=skill_name)) - example_script.chmod(0o755) - print("[OK] Created scripts/example.py") - else: - print("[OK] Created scripts/") - elif resource == "references": - if include_examples: - example_reference = resource_dir / "api_reference.md" - example_reference.write_text(EXAMPLE_REFERENCE.format(skill_title=skill_title)) - print("[OK] Created references/api_reference.md") - else: - print("[OK] Created references/") - elif resource == "assets": - if include_examples: - example_asset = resource_dir / "example_asset.txt" - example_asset.write_text(EXAMPLE_ASSET) - print("[OK] Created assets/example_asset.txt") - else: - print("[OK] Created assets/") - - -def init_skill(skill_name, path, resources, include_examples, interface_overrides): - """ - Initialize a new skill directory with template SKILL.md. - - Args: - skill_name: Name of the skill - path: Path where the skill directory should be created - resources: Resource directories to create - include_examples: Whether to create example files in resource directories - - Returns: - Path to created skill directory, or None if error - """ - # Determine skill directory path - skill_dir = Path(path).resolve() / skill_name - - # Check if directory already exists - if skill_dir.exists(): - print(f"[ERROR] Skill directory already exists: {skill_dir}") - return None - - # Create skill directory - try: - skill_dir.mkdir(parents=True, exist_ok=False) - print(f"[OK] Created skill directory: {skill_dir}") - except Exception as e: - print(f"[ERROR] Error creating directory: {e}") - return None - - # Create SKILL.md from template - skill_title = title_case_skill_name(skill_name) - skill_content = SKILL_TEMPLATE.format(skill_name=skill_name, skill_title=skill_title) - - skill_md_path = skill_dir / "SKILL.md" - try: - skill_md_path.write_text(skill_content) - print("[OK] Created SKILL.md") - except Exception as e: - print(f"[ERROR] Error creating SKILL.md: {e}") - return None - - # Create agents/openai.yaml - try: - result = write_openai_yaml(skill_dir, skill_name, interface_overrides) - if not result: - return None - except Exception as e: - print(f"[ERROR] Error creating agents/openai.yaml: {e}") - return None - - # Create resource directories if requested - if resources: - try: - create_resource_dirs(skill_dir, skill_name, skill_title, resources, include_examples) - except Exception as e: - print(f"[ERROR] Error creating resource directories: {e}") - return None - - # Print next steps - print(f"\n[OK] Skill '{skill_name}' initialized successfully at {skill_dir}") - print("\nNext steps:") - print("1. Edit SKILL.md to complete the TODO items and update the description") - if resources: - if include_examples: - print("2. Customize or delete the example files in scripts/, references/, and assets/") - else: - print("2. Add resources to scripts/, references/, and assets/ as needed") - else: - print("2. Create resource directories only if needed (scripts/, references/, assets/)") - print("3. Update agents/openai.yaml if the UI metadata should differ") - print("4. Run the validator when ready to check the skill structure") - - return skill_dir - - -def main(): - parser = argparse.ArgumentParser( - description="Create a new skill directory with a SKILL.md template.", - ) - parser.add_argument("skill_name", help="Skill name (normalized to hyphen-case)") - parser.add_argument("--path", required=True, help="Output directory for the skill") - parser.add_argument( - "--resources", - default="", - help="Comma-separated list: scripts,references,assets", - ) - parser.add_argument( - "--examples", - action="store_true", - help="Create example files inside the selected resource directories", - ) - parser.add_argument( - "--interface", - action="append", - default=[], - help="Interface override in key=value format (repeatable)", - ) - args = parser.parse_args() - - raw_skill_name = args.skill_name - skill_name = normalize_skill_name(raw_skill_name) - if not skill_name: - print("[ERROR] Skill name must include at least one letter or digit.") - sys.exit(1) - if len(skill_name) > MAX_SKILL_NAME_LENGTH: - print( - f"[ERROR] Skill name '{skill_name}' is too long ({len(skill_name)} characters). " - f"Maximum is {MAX_SKILL_NAME_LENGTH} characters." - ) - sys.exit(1) - if skill_name != raw_skill_name: - print(f"Note: Normalized skill name from '{raw_skill_name}' to '{skill_name}'.") - - resources = parse_resources(args.resources) - if args.examples and not resources: - print("[ERROR] --examples requires --resources to be set.") - sys.exit(1) - - path = args.path - - print(f"Initializing skill: {skill_name}") - print(f" Location: {path}") - if resources: - print(f" Resources: {', '.join(resources)}") - if args.examples: - print(" Examples: enabled") - else: - print(" Resources: none (create as needed)") - print() - - result = init_skill(skill_name, path, resources, args.examples, args.interface) - - if result: - sys.exit(0) - else: - sys.exit(1) - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-skill-creator/scripts/quick_validate.py b/.github/skills/openai-skill-creator/scripts/quick_validate.py deleted file mode 100644 index 0547b404..00000000 --- a/.github/skills/openai-skill-creator/scripts/quick_validate.py +++ /dev/null @@ -1,101 +0,0 @@ -#!/usr/bin/env python3 -""" -Quick validation script for skills - minimal version -""" - -import re -import sys -from pathlib import Path - -import yaml - -MAX_SKILL_NAME_LENGTH = 64 - - -def validate_skill(skill_path): - """Basic validation of a skill""" - skill_path = Path(skill_path) - - skill_md = skill_path / "SKILL.md" - if not skill_md.exists(): - return False, "SKILL.md not found" - - content = skill_md.read_text() - if not content.startswith("---"): - return False, "No YAML frontmatter found" - - match = re.match(r"^---\n(.*?)\n---", content, re.DOTALL) - if not match: - return False, "Invalid frontmatter format" - - frontmatter_text = match.group(1) - - try: - frontmatter = yaml.safe_load(frontmatter_text) - if not isinstance(frontmatter, dict): - return False, "Frontmatter must be a YAML dictionary" - except yaml.YAMLError as e: - return False, f"Invalid YAML in frontmatter: {e}" - - allowed_properties = {"name", "description", "license", "allowed-tools", "metadata"} - - unexpected_keys = set(frontmatter.keys()) - allowed_properties - if unexpected_keys: - allowed = ", ".join(sorted(allowed_properties)) - unexpected = ", ".join(sorted(unexpected_keys)) - return ( - False, - f"Unexpected key(s) in SKILL.md frontmatter: {unexpected}. Allowed properties are: {allowed}", - ) - - if "name" not in frontmatter: - return False, "Missing 'name' in frontmatter" - if "description" not in frontmatter: - return False, "Missing 'description' in frontmatter" - - name = frontmatter.get("name", "") - if not isinstance(name, str): - return False, f"Name must be a string, got {type(name).__name__}" - name = name.strip() - if name: - if not re.match(r"^[a-z0-9-]+$", name): - return ( - False, - f"Name '{name}' should be hyphen-case (lowercase letters, digits, and hyphens only)", - ) - if name.startswith("-") or name.endswith("-") or "--" in name: - return ( - False, - f"Name '{name}' cannot start/end with hyphen or contain consecutive hyphens", - ) - if len(name) > MAX_SKILL_NAME_LENGTH: - return ( - False, - f"Name is too long ({len(name)} characters). " - f"Maximum is {MAX_SKILL_NAME_LENGTH} characters.", - ) - - description = frontmatter.get("description", "") - if not isinstance(description, str): - return False, f"Description must be a string, got {type(description).__name__}" - description = description.strip() - if description: - if "<" in description or ">" in description: - return False, "Description cannot contain angle brackets (< or >)" - if len(description) > 1024: - return ( - False, - f"Description is too long ({len(description)} characters). Maximum is 1024 characters.", - ) - - return True, "Skill is valid!" - - -if __name__ == "__main__": - if len(sys.argv) != 2: - print("Usage: python quick_validate.py ") - sys.exit(1) - - valid, message = validate_skill(sys.argv[1]) - print(message) - sys.exit(0 if valid else 1) diff --git a/.github/skills/openai-slides/LICENSE.txt b/.github/skills/openai-slides/LICENSE.txt deleted file mode 100644 index b5ed4ece..00000000 --- a/.github/skills/openai-slides/LICENSE.txt +++ /dev/null @@ -1,201 +0,0 @@ - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - - Copyright (c) Microsoft Corporation. - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. diff --git a/.github/skills/openai-slides/SKILL.md b/.github/skills/openai-slides/SKILL.md deleted file mode 100644 index cfce1281..00000000 --- a/.github/skills/openai-slides/SKILL.md +++ /dev/null @@ -1,71 +0,0 @@ ---- -name: openai-slides -description: Create and edit presentation slide decks (`.pptx`) with PptxGenJS, bundled layout helpers, and render/validation utilities. Use when tasks involve building a new PowerPoint deck, recreating slides from screenshots/PDFs/reference decks, modifying slide content while preserving editable output, adding charts/diagrams/visuals, or diagnosing layout issues such as overflow, overlaps, and font substitution. ---- - -# Slides - -## Overview - -Use PptxGenJS for slide authoring. Do not use `python-pptx` for deck generation unless the task is inspection-only; keep editable output in JavaScript and deliver both the `.pptx` and the source `.js`. - -Keep work in a task-local directory. Only copy final artifacts to the requested destination after rendering and validation pass. - -## Bundled Resources - -- `assets/pptxgenjs_helpers/`: Copy this folder into the deck workspace and import it locally instead of reimplementing helper logic. -- `scripts/render_slides.py`: Rasterize a `.pptx` or `.pdf` to per-slide PNGs. -- `scripts/slides_test.py`: Detect content that overflows the slide canvas. -- `scripts/create_montage.py`: Build a contact-sheet style montage of rendered slides. -- `scripts/detect_font.py`: Report missing or substituted fonts as LibreOffice resolves them. -- `scripts/ensure_raster_image.py`: Convert SVG/EMF/HEIC/PDF-like assets into PNGs for quick inspection. -- `references/pptxgenjs-helpers.md`: Load only when you need API details or dependency notes. - -## Workflow - -1. Inspect the request and determine whether you are creating a new deck, recreating an existing deck, or editing one. -2. Set the slide size up front. Default to 16:9 (`LAYOUT_WIDE`) unless the source material clearly uses another aspect ratio. -3. Copy `assets/pptxgenjs_helpers/` into the working directory and import the helpers from there. -4. Build the deck in JavaScript with an explicit theme font, stable spacing, and editable PowerPoint-native elements when practical. -5. Run the bundled scripts from this skill directory or copy the needed ones into the task workspace. Render the result with `render_slides.py`, review the PNGs, and fix layout issues before delivery. -6. Run `slides_test.py` for overflow checks when slide edges are tight or the deck is dense. -7. Deliver the `.pptx`, the authoring `.js`, and any generated assets that are required to rebuild the deck. - -## Authoring Rules - -- Set theme fonts explicitly. Do not rely on PowerPoint defaults if typography matters. -- Use `autoFontSize`, `calcTextBox`, and related helpers to size text boxes; do not use PptxGenJS `fit` or `autoFit`. -- Use bullet options, not literal `•` characters. -- Use `imageSizingCrop` or `imageSizingContain` instead of PptxGenJS built-in image sizing. -- Use `latexToSvgDataUri()` for equations and `codeToRuns()` for syntax-highlighted code blocks. -- Prefer native PowerPoint charts for simple bar/line/pie/histogram style visuals so reviewers can edit them later. -- For charts or diagrams that PptxGenJS cannot express well, render SVG externally and place the SVG in the slide. -- Include both `warnIfSlideHasOverlaps(slide, pptx)` and `warnIfSlideElementsOutOfBounds(slide, pptx)` in the submitted JavaScript whenever you generate or substantially edit slides. -- Fix all unintentional overlap and out-of-bounds warnings before delivering. If an overlap is intentional, leave a short code comment near the relevant element. - -## Recreate Or Edit Existing Slides - -- Render the source deck or reference PDF first so you can compare slide geometry visually. -- Match the original aspect ratio before rebuilding layout. -- Preserve editability where possible: text should stay text, and simple charts should stay native charts. -- If a reference slide uses raster artwork, use `ensure_raster_image.py` to generate debug PNGs from vector or odd image formats before placing them. - -## Validation Commands - -Examples below assume you copied the needed scripts into the working directory. If not, invoke the same script paths relative to this skill folder. - -```bash -# Render slides to PNGs for review -python3 scripts/render_slides.py deck.pptx --output_dir rendered - -# Build a montage for quick scanning -python3 scripts/create_montage.py --input_dir rendered --output_file montage.png - -# Check for overflow beyond the original slide canvas -python3 scripts/slides_test.py deck.pptx - -# Detect missing or substituted fonts -python3 scripts/detect_font.py deck.pptx --json -``` - -Load `references/pptxgenjs-helpers.md` if you need the helper API summary or dependency details. diff --git a/.github/skills/openai-slides/agents/openai.yaml b/.github/skills/openai-slides/agents/openai.yaml deleted file mode 100644 index 6f77e499..00000000 --- a/.github/skills/openai-slides/agents/openai.yaml +++ /dev/null @@ -1,6 +0,0 @@ -interface: - display_name: "Slides" - short_description: "Create and edit PPTX slide decks" - icon_small: "./assets/slides-small.svg" - icon_large: "./assets/slides.png" - default_prompt: "Use $slides to create or update this PPTX slide deck with PptxGenJS and validate the layout." diff --git a/.github/skills/openai-slides/assets/pptxgenjs_helpers/code.js b/.github/skills/openai-slides/assets/pptxgenjs_helpers/code.js deleted file mode 100644 index 951ce13a..00000000 --- a/.github/skills/openai-slides/assets/pptxgenjs_helpers/code.js +++ /dev/null @@ -1,104 +0,0 @@ -// Copyright (c) OpenAI. All rights reserved. -"use strict"; - -const fs = require("fs"); -const Prism = require("prismjs"); -let THEME_MAP; - -function loadPrismLanguage(lang) { - const normalized = String(lang || "plaintext").toLowerCase(); - const known = new Set([ - "markup", - "html", - "xml", - "svg", - "mathml", - "css", - "clike", - "javascript", - "js", - "typescript", - "ts", - "python", - "py", - "bash", - "sh", - "json", - "yaml", - "yml", - ]); - const map = { - js: "javascript", - ts: "typescript", - py: "python", - sh: "bash", - yml: "yaml", - html: "markup", - xml: "markup", - }; - const id = map[normalized] || normalized; - if (!Prism.languages[id]) { - try { - require(`prismjs/components/prism-${id}`); - } catch (_e) {} - } - return Prism.languages[id] || Prism.languages.plain || {}; -} - -function buildThemeMap(themeCssModule = "prismjs/themes/prism-okaidia.css") { - try { - const css = fs.readFileSync(require.resolve(themeCssModule), "utf8"); - return Object.fromEntries( - [ - ...css.matchAll( - /\.token\.([\w-]+)[^{]*\{[^}]*color:\s*([^;\s]+)[^}]*\}/g - ), - ].map(([, t, c]) => [t, c.replace(/#|!important/g, "").trim()]) - ); - } catch (err) { - return { plain: "FFFFFF", comment: "999999" }; - } -} - -function getThemeMap() { - if (!THEME_MAP) THEME_MAP = buildThemeMap(); - return THEME_MAP; -} - -function run(text, type = "plain") { - const theme = getThemeMap(); - return { - text, - options: { - fontFace: "Consolas", - color: theme[type] || theme.plain || "FFFFFF", - fontSize: 14, - }, - }; -} - -function tokensToRuns(tokens) { - return tokens.flatMap((t) => - typeof t === "string" - ? [run(t)] - : Array.isArray(t.content) - ? tokensToRuns(t.content) - : [run(t.content, t.type)] - ); -} - -function codeToRuns(code, lang) { - const grammar = loadPrismLanguage(lang); - const lines = String(code || "").split("\n"); - const pad = lines.length.toString().length; - return lines.flatMap((line, i) => [ - run(`${(i + 1).toString().padStart(pad, " ")} `, "comment"), - ...tokensToRuns(Prism.tokenize(line, grammar)), - ...(i < lines.length - 1 ? [run("\n")] : []), - ]); -} - -module.exports = { - codeToRuns, - buildThemeMap, -}; diff --git a/.github/skills/openai-slides/assets/pptxgenjs_helpers/image.js b/.github/skills/openai-slides/assets/pptxgenjs_helpers/image.js deleted file mode 100644 index 16fdc16a..00000000 --- a/.github/skills/openai-slides/assets/pptxgenjs_helpers/image.js +++ /dev/null @@ -1,333 +0,0 @@ -// Copyright (c) OpenAI. All rights reserved. -"use strict"; - -const fs = require("fs"); - -// Accept either a filesystem path, a data URI, raw SVG string, or a Buffer -// and normalize to a Buffer for type/size probing. -function readInputAsBuffer(source) { - if (!source) throw new Error("Image source is empty"); - if (Buffer.isBuffer(source)) return { buffer: source, type: "buffer" }; - if (typeof source === "string") { - // data URI (we primarily emit base64 data URIs for SVG via helpers) - if (source.startsWith("data:")) { - const type = "dataUri"; - const comma = source.indexOf(","); - const payload = comma !== -1 ? source.slice(comma + 1) : source; - // Our helpers use base64; if not, try URI decode then treat as raw text - try { - return { buffer: Buffer.from(payload, "base64"), type: type }; - } catch (_e) { - try { - return { - buffer: Buffer.from(decodeURIComponent(payload), "utf8"), - type: type, - }; - } catch (_e2) { - return { buffer: Buffer.from(payload, "utf8"), type: type }; - } - } - } - // Raw inline SVG string - if (source.includes("= 24 && - buf[0] === 0x89 && - buf[1] === 0x50 && - buf[2] === 0x4e && - buf[3] === 0x47 && - buf[4] === 0x0d && - buf[5] === 0x0a && - buf[6] === 0x1a && - buf[7] === 0x0a - ); -} - -function isJpeg(buf) { - return ( - buf.length > 3 && buf[0] === 0xff && buf[1] === 0xd8 && buf[2] === 0xff - ); -} - -function isGif(buf) { - return ( - buf.length >= 10 && - buf[0] === 0x47 && - buf[1] === 0x49 && - buf[2] === 0x46 && - buf[3] === 0x38 && - (buf[4] === 0x39 || buf[4] === 0x37) && - buf[5] === 0x61 - ); -} - -function isWebp(buf) { - return ( - buf.length >= 16 && - buf[0] === 0x52 && - buf[1] === 0x49 && - buf[2] === 0x46 && - buf[3] === 0x46 && - buf[8] === 0x57 && - buf[9] === 0x45 && - buf[10] === 0x42 && - buf[11] === 0x50 - ); -} - -function isSvg(buf) { - const head = buf.slice(0, 200).toString("utf8"); - return head.includes("> 6)); - return { width, height, type: "webp" }; - } - } - offset += 8 + ((chunkSize + 1) & ~1); // chunks are padded to even size - } - throw new Error("Unsupported WEBP variant for size detection"); -} - -function readJpegSize(buf) { - let offset = 2; - while (offset < buf.length) { - if (buf[offset] !== 0xff) { - offset++; - continue; - } - const marker = buf[offset + 1]; - // SOF0..SOF3, SOF5..SOF7, SOF9..SOF11, SOF13..SOF15 - if ( - (marker >= 0xc0 && marker <= 0xc3) || - (marker >= 0xc5 && marker <= 0xc7) || - (marker >= 0xc9 && marker <= 0xcb) || - (marker >= 0xcd && marker <= 0xcf) - ) { - const blockLength = buf.readUInt16BE(offset + 2); - const height = buf.readUInt16BE(offset + 5); - const width = buf.readUInt16BE(offset + 7); - return { width, height, type: "jpeg" }; - } - const blockLength = buf.readUInt16BE(offset + 2); - if (!Number.isFinite(blockLength) || blockLength < 2) break; - offset += 2 + blockLength; - } - throw new Error("JPEG size not found"); -} - -function parseSvgSize(buf) { - const text = buf.toString("utf8"); - const a = text.indexOf(""); - const inner = a !== -1 && b !== -1 ? text.slice(a, b + 6) : text; - const widthMatch = inner.match(/\bwidth\s*=\s*"([^"]+)"/i); - const heightMatch = inner.match(/\bheight\s*=\s*"([^"]+)"/i); - const viewBoxMatch = inner.match(/\bviewBox\s*=\s*"([^"]+)"/i); - - function toPx(v) { - if (!v) return null; - const m = String(v) - .trim() - .match(/([0-9.]+)\s*(px|pt|em|ex|cm|mm|in|%)?/i); - if (!m) return null; - const n = parseFloat(m[1]); - const unit = (m[2] || "px").toLowerCase(); - const dpi = 96; - switch (unit) { - case "px": - return n; - case "pt": - return (n * dpi) / 72; - case "in": - return n * dpi; - case "cm": - return (n * dpi) / 2.54; - case "mm": - return (n * dpi) / 25.4; - case "em": - case "ex": - return n * 16; // rough fallback - default: - return null; - } - } - - let widthPx = widthMatch ? toPx(widthMatch[1]) : null; - let heightPx = heightMatch ? toPx(heightMatch[1]) : null; - if ((widthPx == null || heightPx == null) && viewBoxMatch) { - const parts = viewBoxMatch[1].trim().split(/\s+/).map(Number); - if (parts.length === 4) { - const vbw = parts[2]; - const vbh = parts[3]; - if (!widthPx && vbh) widthPx = vbw; - if (!heightPx && vbw) heightPx = vbh; - } - } - if (!widthPx || !heightPx) { - // Fallback if sizes missing - widthPx = widthPx || 100; - heightPx = heightPx || 100; - } - return { width: widthPx, height: heightPx, type: "svg" }; -} - -function getImageDimensions(pathOrData) { - const { buffer: buf, type } = readInputAsBuffer(pathOrData); - let meta; - if (isPng(buf)) meta = readPngSize(buf); - else if (isJpeg(buf)) meta = readJpegSize(buf); - else if (isGif(buf)) meta = readGifSize(buf); - else if (isWebp(buf)) meta = readWebpSize(buf); - else if (isSvg(buf)) meta = parseSvgSize(buf); - else { - const suffix = - type === "path" && typeof pathOrData === "string" - ? ` (path: ${pathOrData})` - : ""; - throw new Error("Unsupported image format for provided source" + suffix); - } - - const aspectRatio = - meta.width > 0 && meta.height > 0 ? meta.width / meta.height : 1; - return { - width: meta.width, - height: meta.height, - aspectRatio, - type: meta.type, - }; -} - -function imageSizingCrop(source, x, y, w, h, cx, cy, cw, ch) { - const { aspectRatio } = getImageDimensions(source); - const boxAspect = w / h; - - if ( - cx === undefined || - cy === undefined || - cw === undefined || - ch === undefined - ) { - let cropXFrac, cropYFrac, cropWFrac, cropHFrac; - if (aspectRatio >= boxAspect) { - cropHFrac = 1; - cropWFrac = boxAspect / aspectRatio; - cropXFrac = (1 - cropWFrac) / 2; - cropYFrac = 0; - } else { - cropWFrac = 1; - cropHFrac = aspectRatio / boxAspect; - cropXFrac = 0; - cropYFrac = (1 - cropHFrac) / 2; - } - cx = cropXFrac; - cy = cropYFrac; - cw = cropWFrac; - ch = cropHFrac; - } - - let virtualW = w / cw; - let virtualH = virtualW / aspectRatio; - const eps = 1e-6; - if (Math.abs(virtualH * ch - h) > eps) { - virtualH = h / ch; - virtualW = virtualH * aspectRatio; - } - - const cropXIn = cx * virtualW; - const cropYIn = cy * virtualH; - return { - x, - y, - w: virtualW, - h: virtualH, - sizing: { - type: "crop", - x: cropXIn, - y: cropYIn, - w: w, - h: h, - }, - }; -} - -function imageSizingContain(source, x, y, w, h) { - const { aspectRatio } = getImageDimensions(source); - let w2, h2; - const boxAspect = w / h; - if (aspectRatio >= boxAspect) { - w2 = w; - h2 = w2 / aspectRatio; - } else { - h2 = h; - w2 = h2 * aspectRatio; - } - return { - x: x + (w - w2) / 2, - y: y + (h - h2) / 2, - w: w2, - h: h2, - }; -} - -module.exports = { - getImageDimensions, - imageSizingCrop, - imageSizingContain, -}; diff --git a/.github/skills/openai-slides/assets/pptxgenjs_helpers/index.js b/.github/skills/openai-slides/assets/pptxgenjs_helpers/index.js deleted file mode 100644 index 07805a29..00000000 --- a/.github/skills/openai-slides/assets/pptxgenjs_helpers/index.js +++ /dev/null @@ -1,33 +0,0 @@ -// Copyright (c) OpenAI. All rights reserved. -"use strict"; - -const VERSION = "1.2.0"; - -const text = require("./text"); -const image = require("./image"); -const svg = require("./svg"); -const latex = require("./latex"); -const code = require("./code"); -const layout = require("./layout"); -const layoutBuilders = require("./layout_builders"); -const util = require("./util"); - -module.exports = { - VERSION, - // text layout - ...text, - // images - ...image, - // svg helpers - ...svg, - // LaTeX -> SVG - ...latex, - // code block -> pptx text runs - ...code, - // slide layout analyzers - ...layout, - // slide layout builders - ...layoutBuilders, - // text layout helpers and utilities - ...util, -}; diff --git a/.github/skills/openai-slides/assets/pptxgenjs_helpers/latex.js b/.github/skills/openai-slides/assets/pptxgenjs_helpers/latex.js deleted file mode 100644 index 03c1e221..00000000 --- a/.github/skills/openai-slides/assets/pptxgenjs_helpers/latex.js +++ /dev/null @@ -1,51 +0,0 @@ -// Copyright (c) OpenAI. All rights reserved. -"use strict"; - -let _mathjax; -let _adaptor; -let _doc; - -function ensureMathJax() { - if (_mathjax && _adaptor && _doc) return; - try { - const { mathjax } = require("mathjax-full/js/mathjax.js"); - const { TeX } = require("mathjax-full/js/input/tex.js"); - const { SVG } = require("mathjax-full/js/output/svg.js"); - const { liteAdaptor } = require("mathjax-full/js/adaptors/liteAdaptor.js"); - const { RegisterHTMLHandler } = require("mathjax-full/js/handlers/html.js"); - const { AllPackages } = require("mathjax-full/js/input/tex/AllPackages.js"); - - _adaptor = liteAdaptor(); - RegisterHTMLHandler(_adaptor); - const tex = new TeX({ packages: AllPackages }); - const out = new SVG({ fontCache: "local" }); - _doc = mathjax.document("", { InputJax: tex, OutputJax: out }); - _mathjax = mathjax; - } catch (err) { - throw new Error( - "mathjax-full is not installed. Run `npm i mathjax-full` or avoid latexToSvgDataUri()." - ); - } -} - -function latexToSvgDataUri(latex, display = true) { - ensureMathJax(); - const html = _adaptor.outerHTML(_doc.convert(latex, { display })); - const a = html.indexOf(""); - let svg = a !== -1 && b !== -1 ? html.slice(a, b + 6) : html; - svg = svg.replace(/<\?xml[^>]*>/g, ""); - if (!/xmlns=\"http:\/\/www\.w3\.org\/2000\/svg\"/.test(svg)) { - svg = svg.replace(/ { - const px = Math.round(parseFloat(num) * 8.5); - return `${attr}="${px}px"`; - }); - svg = svg.replace(/currentColor/g, "#000000"); - return "data:image/svg+xml;base64," + Buffer.from(svg).toString("base64"); -} - -module.exports = { - latexToSvgDataUri, -}; diff --git a/.github/skills/openai-slides/assets/pptxgenjs_helpers/layout.js b/.github/skills/openai-slides/assets/pptxgenjs_helpers/layout.js deleted file mode 100644 index 2ce70f38..00000000 --- a/.github/skills/openai-slides/assets/pptxgenjs_helpers/layout.js +++ /dev/null @@ -1,643 +0,0 @@ -// Copyright (c) OpenAI. All rights reserved. -"use strict"; - -function inferElementType(obj) { - if (!obj) return "unknown"; - const data = obj.data || obj.options || {}; - // Distinguish lines explicitly via type only. Many objects have a 'line' style; don't misclassify those. - if (obj.type === "line") return "line"; - if (obj.type && typeof obj.type === "string") return obj.type; - if (obj.text || typeof data.text === "string") return "text"; - if (data.path || obj.image) return "image"; - if (data.chartType) return "chart"; - if (data.shape || data.line) return "shape"; - if (data.mediaType) return "media"; - if (data.table || Array.isArray(data.rows)) return "table"; - if (data.smartArt) return "smartart"; - return "unknown"; -} - -const TEXT_OVERLAP_ERROR_THRESHOLD = 0.1; -const RECTIFY_DIRECTION_EQUALITY_TOLERANCE = 0.15; - -function warnIfSlideHasOverlaps(slide, pptx, options = {}) { - if (!slide || !Array.isArray(slide._slideObjects)) { - console.warn("Invalid slide object passed to warnIfSlideOverlaps()"); - return; - } - const opts = { - // By default, containment cases are very common (e.g., full-slide backgrounds) - // and usually not actionable. Mute them unless explicitly requested. - muteContainment: - options.muteContainment !== undefined ? options.muteContainment : true, - // Do NOT ignore lines or decorative shapes by default; users want true overlaps. - ignoreLines: - options.ignoreLines !== undefined ? options.ignoreLines : false, - ignoreDecorativeShapes: - options.ignoreDecorativeShapes !== undefined - ? options.ignoreDecorativeShapes - : false, - }; - const slideIndex = - pptx && Array.isArray(pptx._slides) ? pptx._slides.indexOf(slide) : -1; - const slideLabel = - slideIndex >= 0 ? `Slide ${slideIndex + 1}` : "(Unknown slide index)"; - const formatElement = (el) => { - const cx = (el.x + el.w / 2).toFixed(3); - const cy = (el.y + el.h / 2).toFixed(3); - return `element ${el.index} (${el.type}, center_x=${cx}, center_y=${cy})`; - }; - const elements = slide._slideObjects.map((obj, i) => { - const { - x = 0, - y = 0, - w = 0, - h = 0, - fill, - line, - } = obj.data || obj.options || {}; - const type = inferElementType(obj); - const isDecorative = (() => { - if (!opts.ignoreDecorativeShapes) return false; - // Border rectangles used as frames: transparent fill (or fully transparent) with a stroke - const transparency = - typeof fill?.transparency === "number" ? fill.transparency : null; - const hasOnlyBorder = !!line && (!fill || transparency !== null); - const fullyTransparent = transparency !== null && transparency >= 99; - return type === "shape" && hasOnlyBorder && fullyTransparent; - })(); - const ignorable = (opts.ignoreLines && type === "line") || isDecorative; - return { index: i, type, x, y, w, h, ignorable }; - }); - let overlapCount = 0; - let containmentCount = 0; - for (let i = 0; i < elements.length; i++) { - const a = elements[i]; - if (a.ignorable) continue; - for (let j = i + 1; j < elements.length; j++) { - const b = elements[j]; - if (b.ignorable) continue; - const comparison = compareElementPosition(slide, a.index, b.index); - if (comparison.relation === "overlapping") { - // Special-case: diagonal line's bounding box overlapping a rectangle is often a false positive. - const EPS = 1e-6; - const getBounds = (e) => ({ - x: e.x, - y: e.y, - x2: e.x + e.w, - y2: e.y + e.h, - }); - const lineRectFalsePositive = (() => { - const oneIsLine = (a.type === "line") ^ (b.type === "line"); - if (!oneIsLine) return false; - const line = a.type === "line" ? a : b; - const rect = a.type === "line" ? b : a; - // If line is diagonal, verify actual segment intersects rect; if not, ignore. - const isDiagonal = line.w > EPS && line.h > EPS; - const lineSeg = { - x1: line.x, - y1: line.y, - x2: line.x + line.w, - y2: line.y + line.h, - }; - const rectB = getBounds(rect); - const pointInRect = (px, py, rb) => - px >= rb.x - EPS && - px <= rb.x2 + EPS && - py >= rb.y - EPS && - py <= rb.y2 + EPS; - const segsIntersect = (p1, p2, q1, q2) => { - const cross = (ax, ay, bx, by) => ax * by - ay * bx; - const d1x = p2.x - p1.x, - d1y = p2.y - p1.y; - const d2x = q2.x - q1.x, - d2y = q2.y - q1.y; - const denom = cross(d1x, d1y, d2x, d2y); - if (Math.abs(denom) < EPS) { - // Parallel: check colinearity and overlapping projections - const crossCol = cross(q1.x - p1.x, q1.y - p1.y, d1x, d1y); - if (Math.abs(crossCol) > EPS) return false; - const proj = (a, b, c) => - Math.min(Math.max(a, b), Math.max(Math.min(a, b), c)); - const overlapX = !( - Math.max(p1.x, p2.x) < Math.min(q1.x, q2.x) - EPS || - Math.max(q1.x, q2.x) < Math.min(p1.x, p2.x) - EPS - ); - const overlapY = !( - Math.max(p1.y, p2.y) < Math.min(q1.y, q2.y) - EPS || - Math.max(q1.y, q2.y) < Math.min(p1.y, p2.y) - EPS - ); - return overlapX && overlapY; - } - const t = cross(q1.x - p1.x, q1.y - p1.y, d2x, d2y) / denom; - const u = cross(q1.x - p1.x, q1.y - p1.y, d1x, d1y) / denom; - return t >= -EPS && t <= 1 + EPS && u >= -EPS && u <= 1 + EPS; - }; - const intersectsRect = (seg, rb) => { - if ( - pointInRect(seg.x1, seg.y1, rb) || - pointInRect(seg.x2, seg.y2, rb) - ) - return true; - const r1 = { x: rb.x, y: rb.y }, - r2 = { x: rb.x2, y: rb.y }, - r3 = { x: rb.x2, y: rb.y2 }, - r4 = { x: rb.x, y: rb.y2 }; - const p1 = { x: seg.x1, y: seg.y1 }, - p2 = { x: seg.x2, y: seg.y2 }; - return ( - segsIntersect(p1, p2, r1, r2) || - segsIntersect(p1, p2, r2, r3) || - segsIntersect(p1, p2, r3, r4) || - segsIntersect(p1, p2, r4, r1) - ); - }; - return isDiagonal && !intersectsRect(lineSeg, rectB); - })(); - if (!lineRectFalsePositive) { - overlapCount++; - - const severeTextOverlap = (() => { - if (!comparison.intersection) return false; - const exceedsThreshold = (element) => - element.type === "text" && - comparison.intersection.w >= TEXT_OVERLAP_ERROR_THRESHOLD && - comparison.intersection.h >= TEXT_OVERLAP_ERROR_THRESHOLD; - return exceedsThreshold(a) || exceedsThreshold(b); - })(); - if (severeTextOverlap) { - const overlapW = comparison.intersection.w; - const overlapH = comparison.intersection.h; - let rectificationSuggestion = ""; - if (overlapW > EPS && overlapH > EPS) { - const maxOverlap = Math.max(overlapW, overlapH); - const diffRatio = Math.abs(overlapW - overlapH) / maxOverlap; - const directions = []; - // Attempt to determine the primary direction of the overlap. This is the direction - // in which the overlap is smaller (and so requires the smallest adjustment to rectify). - if (diffRatio <= RECTIFY_DIRECTION_EQUALITY_TOLERANCE) { - directions.push("horizontally", "vertically"); - } else if (overlapW < overlapH) { - directions.push("horizontally"); - } else { - directions.push("vertically"); - } - rectificationSuggestion = `Suggestion: reposition elements ${directions.join( - " and " - )}.`; - } - - console.error( - `❌ ${slideLabel}: Severe text overlap detected between ${formatElement( - a - )} and ${formatElement( - b - )} (overlap_horizontal=${comparison.intersection.w.toFixed( - 3 - )}, overlap_vertical=${comparison.intersection.h.toFixed( - 3 - )}). THIS MUST BE FIXED. ${rectificationSuggestion}` - ); - } else { - console.warn( - `⚠️ ${slideLabel}: Overlap detected between ${formatElement( - a - )} and ${formatElement(b)}.` - ); - } - } - } else if (comparison.relation === "contained") { - if (!opts.muteContainment) { - containmentCount++; - const container = elements[comparison.containerIndex]; - const contained = elements[comparison.containedIndex]; - console.warn( - `⚠️ ${slideLabel}: ${formatElement( - contained - )} is fully contained within ${formatElement(container)}` - ); - } else { - // Still count internally when muted? We keep for summary only when un-muted - } - } - } - } - if (!(overlapCount === 0 && (!containmentCount || opts.muteContainment))) { - const issues = []; - if (overlapCount > 0) issues.push(`${overlapCount} overlapping pair(s)`); - if (!opts.muteContainment && containmentCount > 0) - issues.push(`${containmentCount} containment case(s)`); - console.log(`⚠️ ${slideLabel}: Found ${issues.join(" and ")}.`); - } -} - -function compareElementPosition(slide, firstIndex, secondIndex) { - if (!slide || !Array.isArray(slide._slideObjects)) { - throw new Error("Invalid slide object passed to compareElementPosition()"); - } - if ( - typeof firstIndex !== "number" || - typeof secondIndex !== "number" || - !Number.isInteger(firstIndex) || - !Number.isInteger(secondIndex) - ) { - throw new Error("Element indices must be integer values."); - } - const elements = slide._slideObjects; - if ( - firstIndex < 0 || - firstIndex >= elements.length || - secondIndex < 0 || - secondIndex >= elements.length - ) { - throw new Error( - "Element index out of bounds for compareElementPosition()." - ); - } - const EPS = 1e-4; - const getBounds = (obj) => { - const source = obj?.data || obj?.options || {}; - let x = typeof source.x === "number" ? source.x : 0; - let y = typeof source.y === "number" ? source.y : 0; - let w = typeof source.w === "number" ? source.w : 0; - let h = typeof source.h === "number" ? source.h : 0; - if (source.sizing && source.sizing.type === "crop") { - if (typeof source.sizing.w === "number") w = source.sizing.w; - if (typeof source.sizing.h === "number") h = source.sizing.h; - } - return { x, y, w, h, x2: x + w, y2: y + h }; - }; - const boundsA = getBounds(elements[firstIndex]); - const boundsB = getBounds(elements[secondIndex]); - const separated = - boundsA.x2 < boundsB.x - EPS || - boundsB.x2 < boundsA.x - EPS || - boundsA.y2 < boundsB.y - EPS || - boundsB.y2 < boundsA.y - EPS; - if (separated) { - return { - relation: "disjoint", - containerIndex: null, - containedIndex: null, - aBounds: boundsA, - bBounds: boundsB, - intersection: null, - }; - } - const aContainsB = - boundsA.x <= boundsB.x + EPS && - boundsA.y <= boundsB.y + EPS && - boundsA.x2 >= boundsB.x2 - EPS && - boundsA.y2 >= boundsB.y2 - EPS; - const bContainsA = - boundsB.x <= boundsA.x + EPS && - boundsB.y <= boundsA.y + EPS && - boundsB.x2 >= boundsA.x2 - EPS && - boundsB.y2 >= boundsA.y2 - EPS; - const ix1 = Math.max(boundsA.x, boundsB.x); - const iy1 = Math.max(boundsA.y, boundsB.y); - const ix2 = Math.min(boundsA.x2, boundsB.x2); - const iy2 = Math.min(boundsA.y2, boundsB.y2); - const intersectionWidth = Math.max(0, ix2 - ix1); - const intersectionHeight = Math.max(0, iy2 - iy1); - const intersection = - intersectionWidth > EPS && intersectionHeight > EPS - ? { x: ix1, y: iy1, w: intersectionWidth, h: intersectionHeight } - : null; - if (aContainsB && !bContainsA) { - return { - relation: "contained", - containerIndex: firstIndex, - containedIndex: secondIndex, - aBounds: boundsA, - bBounds: boundsB, - intersection, - }; - } - if (bContainsA && !aContainsB) { - return { - relation: "contained", - containerIndex: secondIndex, - containedIndex: firstIndex, - aBounds: boundsA, - bBounds: boundsB, - intersection, - }; - } - if (intersection) { - return { - relation: "overlapping", - containerIndex: null, - containedIndex: null, - aBounds: boundsA, - bBounds: boundsB, - intersection, - }; - } - return { - relation: "touching", - containerIndex: null, - containedIndex: null, - aBounds: boundsA, - bBounds: boundsB, - intersection: null, - }; -} - -const VALID_ALIGNMENTS = new Set([ - "left", - "right", - "top", - "bottom", - "verticallyCenter", - "horizontallyCenter", -]); - -const getElementBounds = (obj) => { - const source = obj?.data || obj?.options || {}; - let x = typeof source.x === "number" ? source.x : 0; - let y = typeof source.y === "number" ? source.y : 0; - let w = typeof source.w === "number" ? source.w : 0; - let h = typeof source.h === "number" ? source.h : 0; - // If an image is placed with crop sizing, pptxgenjs stores a larger virtual image w/h - // and a viewport in source.sizing.{w,h}. For visual overlap purposes, use the viewport. - if (source.sizing && source.sizing.type === "crop") { - if (typeof source.sizing.w === "number") w = source.sizing.w; - if (typeof source.sizing.h === "number") h = source.sizing.h; - } - return { x, y, w, h, x2: x + w, y2: y + h }; -}; - -const setElementPosition = (obj, coords) => { - const ensureTarget = (targetObj) => { - if (!targetObj || typeof targetObj !== "object") return null; - return targetObj; - }; - const targets = []; - const dataTarget = ensureTarget(obj.data); - if (dataTarget) targets.push(dataTarget); - const optionsTarget = - obj.options && obj.options !== obj.data ? ensureTarget(obj.options) : null; - if (optionsTarget) targets.push(optionsTarget); - if (targets.length === 0) { - obj.data = obj.data && typeof obj.data === "object" ? obj.data : {}; - targets.push(obj.data); - } - targets.forEach((target) => { - if (coords.x !== undefined) target.x = coords.x; - if (coords.y !== undefined) target.y = coords.y; - }); -}; - -const dimensionKeyPairs = [ - ["width", "height"], - ["w", "h"], - ["cx", "cy"], - ["slideWidth", "slideHeight"], - ["slideWidthInches", "slideHeightInches"], - ["widthInches", "heightInches"], -]; - -const toNumber = (value) => { - if (typeof value === "number" && Number.isFinite(value)) return value; - if (typeof value === "string") { - const parsed = parseFloat(value); - return Number.isFinite(parsed) ? parsed : null; - } - return null; -}; - -const readDimensionsFromObject = (candidate, seen = new Set()) => { - if (!candidate || typeof candidate !== "object") return null; - if (seen.has(candidate)) return null; - seen.add(candidate); - for (const [wKey, hKey] of dimensionKeyPairs) { - const width = toNumber(candidate[wKey]); - const height = toNumber(candidate[hKey]); - if (width !== null && height !== null && width > 0 && height > 0) { - return { width, height }; - } - } - const nestedKeys = ["size", "slideSize", "layout", "slideLayout"]; - for (const key of nestedKeys) { - const nested = readDimensionsFromObject(candidate[key], seen); - if (nested) return nested; - } - return null; -}; - -const getSlideDimensions = (slide, pptx) => { - const candidates = [ - slide?._presLayout, - slide?._slideLayout, - slide?._pres?.layout, - slide?._parent?.layout, - slide?._layout, - pptx?._presLayout, - pptx?._layout, - pptx?.layout, - pptx?.presLayout, - ]; - for (const candidate of candidates) { - const dims = readDimensionsFromObject(candidate); - if (dims) { - // Some internals are in EMUs; convert if values look too large for inches - const EMU_PER_IN = 914400; - const looksEmu = dims.width > 1000 || dims.height > 1000; - if (looksEmu) { - return { - width: dims.width / EMU_PER_IN, - height: dims.height / EMU_PER_IN, - source: "emu_converted", - }; - } - return { ...dims, source: "detected" }; - } - } - throw new Error( - "getSlideDimensions(): Unable to determine slide dimensions from pptxgenjs internals." - ); -}; - -function alignSlideElements(slide, indices, alignment) { - if (!slide || !Array.isArray(slide._slideObjects)) { - throw new Error("Invalid slide object passed to alignSlideElements()"); - } - if (!Array.isArray(indices) || indices.length === 0) { - throw new Error("indices must be a non-empty array."); - } - if (!VALID_ALIGNMENTS.has(alignment)) { - throw new Error(`Unsupported alignment option: ${alignment}`); - } - const uniqueIndices = [...new Set(indices)]; - const elements = slide._slideObjects; - const selected = uniqueIndices.map((idx) => { - if (typeof idx !== "number" || !Number.isInteger(idx)) { - throw new Error("Element indices must be integers."); - } - if (idx < 0 || idx >= elements.length) { - throw new Error("Element index out of bounds for alignSlideElements()."); - } - const obj = elements[idx]; - const bounds = getElementBounds(obj); - return { index: idx, obj, bounds }; - }); - if (selected.length < 2) return; - const minX = Math.min(...selected.map((item) => item.bounds.x)); - const maxX2 = Math.max(...selected.map((item) => item.bounds.x2)); - const minY = Math.min(...selected.map((item) => item.bounds.y)); - const maxY2 = Math.max(...selected.map((item) => item.bounds.y2)); - const centerX = (minX + maxX2) / 2; - const centerY = (minY + maxY2) / 2; - selected.forEach(({ obj, bounds }) => { - const { w, h } = bounds; - switch (alignment) { - case "left": - setElementPosition(obj, { x: minX }); - break; - case "right": - setElementPosition(obj, { x: maxX2 - w }); - break; - case "top": - setElementPosition(obj, { y: minY }); - break; - case "bottom": - setElementPosition(obj, { y: maxY2 - h }); - break; - case "horizontallyCenter": - setElementPosition(obj, { x: centerX - w / 2 }); - break; - case "verticallyCenter": - setElementPosition(obj, { y: centerY - h / 2 }); - break; - default: - throw new Error(`Unhandled alignment option: ${alignment}`); - } - }); -} - -function distributeSlideElements(slide, indices, direction) { - if (!slide || !Array.isArray(slide._slideObjects)) { - throw new Error("Invalid slide object passed to distributeSlideElements()"); - } - if (!Array.isArray(indices) || indices.length === 0) { - throw new Error("indices must be a non-empty array."); - } - if (direction !== "horizontal" && direction !== "vertical") { - throw new Error(`Unsupported distribution direction: ${direction}`); - } - const uniqueIndices = [...new Set(indices)]; - if (uniqueIndices.length < 2) return; - const elements = slide._slideObjects; - const selected = uniqueIndices.map((idx) => { - if (typeof idx !== "number" || !Number.isInteger(idx)) { - throw new Error("Element indices must be integers."); - } - if (idx < 0 || idx >= elements.length) { - throw new Error( - "Element index out of bounds for distributeSlideElements()." - ); - } - const obj = elements[idx]; - const bounds = getElementBounds(obj); - return { index: idx, obj, bounds }; - }); - const axisStartKey = direction === "horizontal" ? "x" : "y"; - const axisEndKey = direction === "horizontal" ? "x2" : "y2"; - const sizeKey = direction === "horizontal" ? "w" : "h"; - selected.sort((a, b) => { - const delta = a.bounds[axisStartKey] - b.bounds[axisStartKey]; - return Math.abs(delta) > 1e-6 ? delta : a.index - b.index; - }); - const minCoord = Math.min( - ...selected.map((item) => item.bounds[axisStartKey]) - ); - const maxCoord = Math.max(...selected.map((item) => item.bounds[axisEndKey])); - const totalSpan = maxCoord - minCoord; - const gaps = selected.length - 1; - const totalSize = selected.reduce( - (sum, item) => sum + item.bounds[sizeKey], - 0 - ); - const gapSize = gaps > 0 ? (totalSpan - totalSize) / gaps : 0; - let cursor = minCoord; - selected.forEach(({ obj, bounds }) => { - if (direction === "horizontal") { - setElementPosition(obj, { x: cursor }); - cursor += bounds.w + gapSize; - } else { - setElementPosition(obj, { y: cursor }); - cursor += bounds.h + gapSize; - } - }); -} - -function warnIfSlideElementsOutOfBounds(slide, pptx) { - if (!slide || !Array.isArray(slide._slideObjects)) { - console.warn( - "Invalid slide object passed to warnIfSlideElementsOutOfBounds()" - ); - return; - } - const { - width: slideWidth, - height: slideHeight, - source, - } = getSlideDimensions(slide, pptx); - const slideIndex = - pptx && Array.isArray(pptx._slides) ? pptx._slides.indexOf(slide) : -1; - const slideLabel = - slideIndex >= 0 ? `Slide ${slideIndex + 1}` : "(Unknown slide index)"; - if (source === "default") { - console.warn( - `⚠️ ${slideLabel}: Unable to determine slide dimensions from pptxgenjs internals; assuming width=${slideWidth}, height=${slideHeight}.` - ); - } - const EPS = 1e-4; - let outOfBoundsCount = 0; - const formatElement = (idx, type, bounds) => { - const cx = (bounds.x + bounds.w / 2).toFixed(3); - const cy = (bounds.y + bounds.h / 2).toFixed(3); - return `Element ${idx} (${type}, center_x=${cx}, center_y=${cy})`; - }; - slide._slideObjects.forEach((obj, index) => { - const bounds = getElementBounds(obj); - const type = inferElementType(obj); - const violations = []; - if (bounds.x < -EPS) violations.push(`left=${bounds.x.toFixed(3)} < 0`); - if (bounds.y < -EPS) violations.push(`top=${bounds.y.toFixed(3)} < 0`); - if (bounds.x2 > slideWidth + EPS) - violations.push( - `right=${bounds.x2.toFixed(3)} > width=${slideWidth.toFixed(3)}` - ); - if (bounds.y2 > slideHeight + EPS) - violations.push( - `bottom=${bounds.y2.toFixed(3)} > height=${slideHeight.toFixed(3)}` - ); - if (violations.length > 0) { - outOfBoundsCount++; - console.warn( - `⚠️ ${slideLabel}: ${formatElement( - index, - type, - bounds - )} exceeds slide bounds (${violations.join(", ")}).` - ); - } - }); - if (outOfBoundsCount > 0) { - console.log( - `⚠️ ${slideLabel}: Found ${outOfBoundsCount} element(s) extending beyond the slide bounds.` - ); - } -} - -module.exports = { - inferElementType, - compareElementPosition, - warnIfSlideHasOverlaps, - alignSlideElements, - distributeSlideElements, - warnIfSlideElementsOutOfBounds, - getSlideDimensions, -}; diff --git a/.github/skills/openai-slides/assets/pptxgenjs_helpers/layout_builders.js b/.github/skills/openai-slides/assets/pptxgenjs_helpers/layout_builders.js deleted file mode 100644 index f26990b6..00000000 --- a/.github/skills/openai-slides/assets/pptxgenjs_helpers/layout_builders.js +++ /dev/null @@ -1,358 +0,0 @@ -// Copyright (c) OpenAI. All rights reserved. -"use strict"; - -const { calcTextBox, autoFontSize } = require("./text"); -const { imageSizingCrop, imageSizingContain } = require("./image"); -const { getSlideDimensions } = require("./layout"); - -module.exports = { - addImageTextCard, - addCardRow, - addThreeLevelTree, -}; - -function addImageTextCard(slide, opts = {}) { - const x = toNumberOr(opts.x, 0); - const y = toNumberOr(opts.y, 0); - const w = toNumberOr(opts.width, 3.0); - const gap = toNumberOr(opts.gap, 0.15); - const image = opts.image || {}; - const text = opts.text || ""; - const textBox = opts.textBox || {}; - - const boxH = toNumberOr(image.boxHeight, 2.2); - const sizing = (image.sizing || "crop").toLowerCase(); - let imgPlacement; - if (image.path || image.data) { - const base = image.path ? { path: image.path } : { data: image.data }; - if (sizing === "contain") { - imgPlacement = imageSizingContain( - image.path || image.data, - x, - y, - w, - boxH - ); - slide.addImage({ ...base, ...imgPlacement }); - } else { - const c = image.crop || {}; - imgPlacement = imageSizingCrop( - image.path || image.data, - x, - y, - w, - boxH, - c.cx, - c.cy, - c.cw, - c.ch - ); - slide.addImage({ ...base, ...imgPlacement }); - } - } - - const textY = y + boxH + gap; - const fontSize = toNumberOr(textBox.fontSize, 14); - const fontFaceRaw = textBox.fontFace; - const fontFace = - typeof fontFaceRaw === "string" && fontFaceRaw.trim().length > 0 - ? fontFaceRaw.trim() - : null; - if (!fontFace) { - throw new Error( - "addImageTextCard(): textBox.fontFace is required for text measurement." - ); - } - let hText; - let textOptions; - - if (textBox.h != null && Number.isFinite(toNumberOr(textBox.h, NaN))) { - // Layout-first: caller fixed the box height, so adjust font size to fit via autoFontSize. - const fixedH = toNumberOr(textBox.h, 0); - const baseOpts = { - x, - y: textY, - w, - h: fixedH, - mode: textBox.mode || "auto", - fontSize, - minFontSize: textBox.minFontSize, - maxFontSize: textBox.maxFontSize, - margin: textBox.margin, - paraSpaceAfter: textBox.paraSpaceAfter, - }; - const autoOpts = autoFontSize(text, fontFace, baseOpts); - hText = fixedH; - textOptions = { - ...autoOpts, - fontFace, - color: textBox.color, - align: textBox.align, - valign: textBox.valign || "top", - fill: opts.background, - }; - } else { - // Content-first: fixed font size, let calcTextBox derive the required height. - const layout = calcTextBox(fontSize, { - text, - w, - fontFace, - margin: textBox.margin, - paraSpaceAfter: textBox.paraSpaceAfter, - }); - hText = layout.h; - textOptions = { - x, - y: textY, - w, - h: hText, - fontFace, - fontSize, - color: textBox.color, - align: textBox.align, - valign: textBox.valign || "top", - paraSpaceAfter: textBox.paraSpaceAfter, - margin: textBox.margin, - fill: opts.background, - }; - } - - slide.addText(text, textOptions); - - return { - x, - y, - w, - image: { - x: imgPlacement?.x ?? x, - y, - w: imgPlacement?.w ?? w, - h: imgPlacement?.h ?? boxH, - }, - text: { x, y: textY, w, h: hText }, - }; -} - -function addCardRow(slide, region, cards = [], options = {}) { - const rx = toNumberOr(region.x, 0.4); - const ry = toNumberOr(region.y, 1.6); - const slideWidth = getSlideDimensions(slide).width; - const rw = toNumberOr(region.w, slideWidth - rx * 2); - const gap = toNumberOr(options.gap, 0.25); - const count = cards.length; - if (count === 0) return []; - - let cardW; - if (options.widthStrategy === "fixed") { - cardW = toNumberOr( - options.cardWidth, - rw / count - (gap * (count - 1)) / count - ); - } else { - cardW = (rw - gap * (count - 1)) / count; - } - - const totalWidth = cardW * count + gap * (count - 1); - const align = options.align || "left"; - const ox = - align === "center" - ? (rw - totalWidth) / 2 - : align === "right" - ? rw - totalWidth - : 0; - - const placements = []; - for (let i = 0; i < count; i++) { - const x = rx + ox + i * (cardW + gap); - placements.push( - addImageTextCard(slide, { ...cards[i], x, y: ry, width: cardW }) - ); - } - return placements; -} - -function addThreeLevelTree(slide, opts = {}) { - const slideWidth = getSlideDimensions(slide).width; - const cx = toNumberOr(opts.centerX, slideWidth / 2); - const topY = toNumberOr(opts.topY, 1.6); - - const rootW = toNumberOr(opts.root?.w, 3.3333333); - const rootH = toNumberOr(opts.root?.h, 0.93333333); - const rootX = cx - rootW / 2; - const rootFontFaceRaw = opts.root?.fontFace; - const rootFontFace = - typeof rootFontFaceRaw === "string" && rootFontFaceRaw.trim().length > 0 - ? rootFontFaceRaw.trim() - : null; - if (!rootFontFace) { - throw new Error( - "addThreeLevelTree(): opts.root.fontFace is required for text measurement." - ); - } - const rootFontSize = toNumberOr(opts.root?.fontSize, 16); - const rootText = opts.root?.text || ""; - const rootTextOpts = autoFontSize(rootText, rootFontFace, { - x: rootX, - y: topY, - w: rootW, - h: rootH, - mode: opts.root?.mode || "shrink", - fontSize: rootFontSize, - minFontSize: opts.root?.minFontSize, - maxFontSize: opts.root?.maxFontSize, - }); - slide.addText(rootText, { - ...rootTextOpts, - align: "center", - valign: "mid", - fontFace: rootFontFace, - color: opts.root?.color || "FFFFFF", - fill: { color: opts.root?.fill || "0B0F1A" }, - line: { color: opts.root?.line || opts.root?.fill || "0B0F1A" }, - }); - - const midLabels = Array.isArray(opts.mid?.labels) ? opts.mid.labels : []; - const midFontFaceRaw = opts.mid?.fontFace; - const midFontFace = - typeof midFontFaceRaw === "string" && midFontFaceRaw.trim().length > 0 - ? midFontFaceRaw.trim() - : null; - if (!midFontFace) { - throw new Error( - "addThreeLevelTree(): opts.mid.fontFace is required for text measurement." - ); - } - let midW = toNumberOr(opts.mid?.w, NaN); - const midH = toNumberOr(opts.mid?.h, rootH); - const midY = toNumberOr(opts.mid?.y, topY + rootH + 1.2); - const requestedSpacing = toNumberOr(opts.mid?.spacing, NaN); // center-to-center distance if provided - const leftRightMargin = toNumberOr(opts.mid?.marginX, 0.6); - const availableRowWidth = slideWidth - leftRightMargin * 2; - const countMid = midLabels.length; - const minGap = 0.4; - if (!Number.isFinite(midW) && Number.isFinite(requestedSpacing)) { - // Derive midW from spacing and available width - const totalSpan = requestedSpacing * (countMid - 1) + 0; // span between first and last centers - const maxW = Math.min(rootW, (availableRowWidth - totalSpan) / countMid); - midW = Math.max(0.8, maxW); - } - if (!Number.isFinite(midW)) { - // Fit equally within available width with minimum gaps - midW = Math.max( - 0.8, - (availableRowWidth - minGap * (countMid - 1)) / countMid - ); - } - // Compute gap to center-group horizontally without overlap - let gap = Math.max( - minGap, - (availableRowWidth - midW * countMid) / Math.max(1, countMid - 1) - ); - const totalWidth = midW * countMid + gap * (countMid - 1); - const startLeft = cx - totalWidth / 2; - for (let i = 0; i < midLabels.length; i++) { - const x = startLeft + i * (midW + gap); - const midText = midLabels[i] || ""; - const midFontSize = toNumberOr(opts.mid?.fontSize, 16); - const midTextOpts = autoFontSize(midText, midFontFace, { - x, - y: midY, - w: midW, - h: midH, - mode: opts.mid?.mode || "shrink", - fontSize: midFontSize, - minFontSize: opts.mid?.minFontSize, - maxFontSize: opts.mid?.maxFontSize, - }); - slide.addText(midText, { - ...midTextOpts, - align: "center", - valign: "mid", - fontFace: midFontFace, - color: opts.mid?.color || "000000", - fill: { color: opts.mid?.fill || "A0BEC2" }, - line: { color: opts.mid?.line || opts.mid?.fill || "A0BEC2" }, - }); - addConnector(slide, cx, topY + rootH, x + midW / 2, midY, opts.line); - } - - const leavesPerMid = Array.isArray(opts.leaf?.labelsPerMid) - ? opts.leaf.labelsPerMid - : []; - const leafFontFaceRaw = opts.leaf?.fontFace; - const leafFontFace = - typeof leafFontFaceRaw === "string" && leafFontFaceRaw.trim().length > 0 - ? leafFontFaceRaw.trim() - : null; - if (!leafFontFace) { - throw new Error( - "addThreeLevelTree(): opts.leaf.fontFace is required for text measurement." - ); - } - const leafW = toNumberOr(opts.leaf?.w, 1.05); - const leafH = toNumberOr(opts.leaf?.h, 1.0666667); - const leafY = toNumberOr(opts.leaf?.y, midY + midH + 1.0); - const minLeafGap = 0.2; - for (let i = 0; i < midLabels.length; i++) { - const xBase = startLeft + i * (midW + gap); - const childLabels = Array.isArray(leavesPerMid[i]) ? leavesPerMid[i] : []; - const childCount = childLabels.length || 3; - // Compute per-mid gap to fit children within midW without overlap - const leafGap = Math.max( - minLeafGap, - (midW - childCount * leafW) / Math.max(1, childCount - 1) - ); - const totalWidth = childCount * leafW + (childCount - 1) * leafGap; - const leftX = xBase + (midW - totalWidth) / 2; - for (let j = 0; j < childCount; j++) { - const x = leftX + j * (leafW + leafGap); - const leafText = childLabels[j] || ""; - const leafFontSize = toNumberOr(opts.leaf?.fontSize, 16); - const leafTextOpts = autoFontSize(leafText, leafFontFace, { - x, - y: leafY, - w: leafW, - h: leafH, - mode: opts.leaf?.mode || "shrink", - fontSize: leafFontSize, - minFontSize: opts.leaf?.minFontSize, - maxFontSize: opts.leaf?.maxFontSize, - }); - slide.addText(leafText, { - ...leafTextOpts, - align: "center", - valign: "mid", - fontFace: leafFontFace, - color: opts.leaf?.color || "000000", - fill: { color: opts.leaf?.fill || "A6C1EE" }, - line: { color: opts.leaf?.line || opts.leaf?.fill || "A6C1EE" }, - }); - addConnector( - slide, - xBase + midW / 2, - midY + midH, - x + leafW / 2, - leafY, - opts.line - ); - } - } -} - -function addConnector(slide, x1, y1, x2, y2, line = {}) { - const x = Math.min(x1, x2); - const y = Math.min(y1, y2); - slide.addShape("line", { - x, - y, - w: Math.abs(x2 - x1), - h: Math.abs(y2 - y1), - line: { color: line.color || "000000", pt: line.pt || 1 }, - flipH: x2 < x1 ? true : undefined, - }); -} - -function toNumberOr(v, fallback) { - const n = typeof v === "string" ? parseFloat(v) : v; - return Number.isFinite(n) ? n : fallback; -} diff --git a/.github/skills/openai-slides/assets/pptxgenjs_helpers/svg.js b/.github/skills/openai-slides/assets/pptxgenjs_helpers/svg.js deleted file mode 100644 index 22079946..00000000 --- a/.github/skills/openai-slides/assets/pptxgenjs_helpers/svg.js +++ /dev/null @@ -1,36 +0,0 @@ -// Copyright (c) OpenAI. All rights reserved. -"use strict"; - -function toDataUri(svg) { - return "data:image/svg+xml;base64," + Buffer.from(svg).toString("base64"); -} - -function sanitizeSvg(svg) { - let inner = svg; - const a = inner.indexOf(""); - if (a !== -1 && b !== -1) inner = inner.slice(a, b + 6); - inner = inner.replace(/<\?xml[^>]*>/g, ""); - if (!/xmlns=\"http:\/\/www\.w3\.org\/2000\/svg\"/.test(inner)) { - inner = inner.replace(/ { - const px = Math.round(parseFloat(num) * 8.5); - return `${attr}="${px}px"`; - } - ); - inner = inner.replace(/currentColor/g, "#000000"); - return inner; -} - -function svgToDataUri(svg) { - return toDataUri(sanitizeSvg(svg)); -} - -module.exports = { - toDataUri, - sanitizeSvg, - svgToDataUri, -}; diff --git a/.github/skills/openai-slides/assets/pptxgenjs_helpers/text.js b/.github/skills/openai-slides/assets/pptxgenjs_helpers/text.js deleted file mode 100644 index 22bd8d09..00000000 --- a/.github/skills/openai-slides/assets/pptxgenjs_helpers/text.js +++ /dev/null @@ -1,789 +0,0 @@ -// Copyright (c) OpenAI. All rights reserved. -"use strict"; - -const { spawnSync } = require("child_process"); -const { Canvas } = require("skia-canvas"); -// Unicode line-break iterator (UAX #14) so we mimic PPT/LibreOffice wrapping rules. -const LineBreaker = require("linebreak"); -const fontkit = require("fontkit"); -const TEXT_MEASURER = getTextMeasurer(); -const registeredFontVariants = new Set(); -const fontPathCache = new Map(); -const fontKitCache = new Map(); - -// Estimate the text box height for a given font size and line count. -// NOTE: This is an analytical approximation, not an exact reproduction of -// PowerPoint/LibreOffice layout. Always verify visually and adjust based on -// actual rendering if precise fit is required. -function calcTextBoxHeightSimple( - fontSize, - lines = 1, - leading = 1.15, - padding = 0.3 -) { - const lineHeightIn = (fontSize / 72) * leading; - return lines * lineHeightIn + padding; -} - -// Compute font size that fits given text within a fixed box. -// NOTE: autoFontSize uses skia-canvas measurement stack to approximate the font size -// that will fit in a given box. Rendering engines may differ slightly, so -// treat the result as an estimate and tweak as needed after visual inspection. -// Signature: -// autoFontSize(textOrRuns, fontFace, opts?) -// - fontFace must be provided as the 2nd positional argument and cannot be in opts. -// - All modes always respect [minFontSize, maxFontSize] as a CLOSED interval when provided. -// Modes: -// - mode: "shrink" => shrink only (search [minFontSize, min(maxFontSize, fontSize)]) -// - mode: "enlarge" => enlarge only (search [max(minFontSize, fontSize), maxFontSize]) -// - mode: "auto" => shrink + enlarge (search [minFontSize, maxFontSize]); fontSize optional. -// In "auto" mode fontSize is not required; when omitted we simply search the whole [minFontSize, maxFontSize] range. -// Returns a cloned options object with computed fontSize. fit: "shrink" is appended only when mode === "shrink". -function autoFontSize(textOrRuns, fontFace, opts = {}) { - const x = toNumber(opts.x, 0); - const y = toNumber(opts.y, 0); - const w = toNumber(opts.w, 0); - const h = toNumber(opts.h, 0); - if (!(w > 0 && h > 0)) throw new Error("autoFontSize(): non-positive w or h"); - - const face = typeof fontFace === "string" ? fontFace.trim() : ""; - if (face.length === 0) { - throw new Error( - "autoFontSize(): fontFace is required as the 2nd positional argument." - ); - } - - // Fast-path: if there is no visible text content, just return the - // (optionally clamped) reference fontSize; there is nothing to fit. - const hasAnyText = - normalizeText(textOrRuns).trim().length > 0 || - (Array.isArray(textOrRuns) && - textOrRuns.some( - (run) => run && typeof run.text === "string" && run.text.trim().length - )); - - const fontStyle = - opts.italic === true || opts.fontStyle === "italic" ? "italic" : "normal"; - const fontWeight = - opts.bold === true || String(opts.fontWeight || "").toLowerCase() === "bold" - ? "bold" - : "normal"; - const leading = toNumber(opts.leading, 1.15) || 1.15; - - const modeRaw = typeof opts.mode === "string" ? opts.mode : "auto"; // 'auto' (default) | 'shrink' | 'enlarge' - const mode = modeRaw.toLowerCase(); - const isShrink = mode === "shrink"; - const isEnlarge = mode === "enlarge"; - const isAuto = mode === "auto"; - - const refPtRaw = toNumber(opts.fontSize, NaN); - const hasRefPt = Number.isFinite(refPtRaw); - const refPt = hasRefPt ? refPtRaw : NaN; - - // Base bounds (closed interval). Defaults: - // - minFontSize: 1pt - // - maxFontSize: 1000pt (unless the caller provided a tighter bound) - let minPt = toNumber(opts.minFontSize, NaN); - let maxPt = toNumber(opts.maxFontSize, NaN); - const userProvidedMax = Number.isFinite(maxPt); - if (!Number.isFinite(minPt)) { - minPt = 1; - } - if (!Number.isFinite(maxPt)) { - maxPt = 1000; - } - - if (isShrink || isEnlarge) { - if (!hasRefPt) { - throw new Error( - "autoFontSize(): mode 'shrink' or 'enlarge' requires fontSize" - ); - } - } - - if (isShrink) { - // Shrink only: never exceed the requested size (and respect maxFontSize). - maxPt = Math.min(maxPt, refPt); - } else if (isEnlarge) { - // Enlarge only: never go below the requested size (and respect minFontSize). - minPt = Math.max(minPt, refPt); - } else if (isAuto && hasRefPt && userProvidedMax) { - // Auto mode with an explicit maxFontSize: honor [minFontSize, maxFontSize] - // as the search band while allowing both shrink and enlarge within it. - } else if (!isAuto) { - throw new Error( - `autoFontSize(): unsupported mode "${modeRaw}", expected "auto" | "shrink" | "enlarge"` - ); - } - - if (!(maxPt > 0 && maxPt >= minPt)) { - throw new Error( - "autoFontSize(): invalid minFontSize/maxFontSize bounds after normalization" - ); - } - - // If there is no actual text, we can skip measurement entirely and just - // clamp the reference size to [minPt, maxPt]. - if (!hasAnyText) { - const chosen = - (hasRefPt && Math.max(minPt, Math.min(maxPt, refPt))) || minPt; - const out = { ...opts, x, y, w, h, fontSize: chosen }; - if (isShrink) out.fit = "shrink"; - return out; - } - - // Search the space of candidate font sizes with a small step and a safety - // bias baked into the fit test: - // - precision: 0.05pt (~1/20pt) so we land very close to the true max-fit. - // - safetyFactor: we require that the calcTextBox()-measured height is - // within a small margin of the caller-provided box height, so that the - // same layout engine used by calcTextBox drives autoFontSize decisions. - const precision = 0.05; // point precision for search (~1/20pt) - const safetyFactor = 0.97; - - let lo = minPt; - let hi = maxPt; - let best = lo; - while (hi - lo > precision) { - const mid = (lo + hi) / 2; - // Delegate measurement to calcTextBox so that autoFontSize and - // calcTextBox share the exact same layout pipeline (paragraph modeling, - // bullet handling, margins, padding, width scaling, etc.). - const layout = calcTextBox(mid, { - text: textOrRuns, - w, - fontFace: face, - fontStyle, - fontWeight, - leading, - margin: opts.margin, - padding: opts.padding, - paraSpaceAfter: opts.paraSpaceAfter, - }); - const fits = layout.h <= h * safetyFactor + 1e-6; - if (fits) { - best = mid; - lo = mid; // try larger - } else { - hi = mid; // shrink - } - } - // Closed interval: clamp to [minPt, maxPt]. - const finalPt = Math.max(minPt, Math.min(maxPt, best)); - - // Pass through all original options, override fontSize and append fit: "shrink" - const out = { ...opts, x, y, w, h, fontSize: finalPt }; - if (isShrink) out.fit = "shrink"; - return out; -} - -// Calculate text box metrics using skia-canvas measurement (lines, height, -// width) for a given font size and text payload. -// NOTE: calcTextBox approximates how many lines and how much space text will -// occupy using our JS measurement pipeline. It is designed to be close to -// PowerPoint/LibreOffice but is not guaranteed pixel-perfect—always adjust -// based on actual slide rendering when precision matters. -// Signature: -// calcTextBox(fontSizePt, opts) -// - fontSizePt: number (points) -// - opts (keywords): { -// text?: string | runs[], -// w?: number (inches), -// h?: number (inches), -// lines?: number, -// fontFace?: string, // required when measuring by width/height with text -// fontStyle?: 'normal' | 'italic', italic?: boolean, -// fontWeight?: 'normal' | 'bold', bold?: boolean, -// leading?: number (line height multiplier, default 1.15), -// padding?: number (inches, default 0.3), -// paraSpaceAfter?: number (points, default 0) -// } -// Modes (auto-detected): -// a) Given lines -> compute height -// b) Given width + text -> compute height and lines -// c) Given height + text -> compute width and lines -// Throws when insufficient info is provided. -function calcTextBox(fontSizePt, opts = {}) { - const textInput = opts.text ?? ""; - const text = normalizeText(textInput || ""); - const face = - typeof opts.fontFace === "string" && opts.fontFace.trim().length > 0 - ? opts.fontFace.trim() - : ""; - const fontStyle = - opts.italic === true || opts.fontStyle === "italic" ? "italic" : "normal"; - const fontWeight = - opts.bold === true || String(opts.fontWeight || "").toLowerCase() === "bold" - ? "bold" - : "normal"; - const leading = toNumber(opts.leading, 1.15) || 1.15; - const padding = toNumber(opts.padding, 0.3); // inches (allow 0) - const paraSpaceAfterPt = toNumber(opts.paraSpaceAfter, 0) || 0; // points - const lineHeightIn = (fontSizePt / 72) * leading; - const margins = normalizeMargins(opts.margin); - const measurer = TEXT_MEASURER; - - const hasLines = Number.isFinite(toNumber(opts.lines, NaN)); - const hasWidth = Number.isFinite(toNumber(opts.w, NaN)); - const hasHeight = Number.isFinite(toNumber(opts.h, NaN)); - const paragraphs = buildParagraphModels(textInput, { - fontSizePt, - // Do not silently substitute a default font here; callers measuring by - // width/height are required to pass an explicit fontFace so that our - // metrics match the actual slide theme. - fontFace: face, - fontStyle, - fontWeight, - leading, - paraSpaceAfterPt, - }); - const hasAnyText = paragraphs.some((p) => p.text.length > 0); - - // Empirical top inset: PPT text frames render a small gutter above the first line - // even with zero margins. Model it as a fraction of the font size so callers can - // visually trim by shifting y up and growing h by the same amount. - const topInsetIn = (fontSizePt / 72) * 0.2; // ~20% of font size (inches) - - if (hasLines) { - // Mode (a): Given lines -> compute height only - const lines = toNumber(opts.lines, 1); - const contentH = Math.max(0, lines * lineHeightIn + padding); - const h = contentH + margins.top + margins.bottom; - const passthrough = buildPassthroughOptions(opts, fontSizePt, margins); - return { - ...passthrough, - w: toNumber(opts.w, NaN) || null, - h, - lines, - contentH, - margins, - topInset: topInsetIn, - }; - } - - if (hasWidth && hasAnyText) { - // Mode (b): Given width + text -> compute height and lines - if (face.length === 0) { - throw new Error( - "calcTextBox(): opts.fontFace is required when measuring by width." - ); - } - const boxW = toNumber(opts.w, 0); - if (!(boxW > 0)) - throw new Error("calcTextBox(): width must be > 0 in mode 'width'"); - const innerW = Math.max(0, boxW - margins.left - margins.right); - const { lines, heightIn } = layoutGivenWidth(paragraphs, innerW); - const contentH = Math.max(0, heightIn + padding); - const h = contentH + margins.top + margins.bottom; - const passthrough = buildPassthroughOptions(opts, fontSizePt, margins); - return { - ...passthrough, - w: boxW, - h, - lines, - contentH, - margins, - topInset: topInsetIn, - }; - } - - if (hasHeight && hasAnyText) { - // Mode (c): Given height + text -> compute minimal width and lines to fit - if (face.length === 0) { - throw new Error( - "calcTextBox(): opts.fontFace is required when measuring by height." - ); - } - const boxH = toNumber(opts.h, 0); - if (!(boxH > 0)) - throw new Error("calcTextBox(): height must be > 0 in mode 'height'"); - const innerH = Math.max(0, boxH - margins.top - margins.bottom); - // Upper bound: single-line width across paragraphs - const singleLineWidth = paragraphs.reduce((mx, p) => { - const width = measureRunWidth(p, p.text) + p.textIndentIn; - return Math.max(mx, width); - }, 0); - const minHeightOneLine = Math.max( - 0, - paragraphs.reduce((sum, p, idx) => { - const lineHeight = (p.fontSizePt / 72) * p.leading; - sum += lineHeight; - if (idx !== paragraphs.length - 1) sum += p.paraSpaceAfterIn; - return sum; - }, 0) - ); - if (minHeightOneLine + padding - innerH > 1e-6) { - throw new Error( - "calcTextBox(): height too small for one-line layout at this font size" - ); - } - // Lower bound: longest token width - const longestTokenWidth = paragraphs.reduce((mx, p) => { - const tokens = splitTextIntoTokens(p.text); - for (const tk of tokens) { - if (tk.length === 0) continue; - const wIn = measureRunWidth(p, tk) + p.textIndentIn; - if (wIn > mx) mx = wIn; - } - return mx; - }, 0); - let lo = Math.max(0.01, longestTokenWidth); - let hi = Math.max(lo, singleLineWidth); - let best = hi; - for (let iter = 0; iter < 32; iter++) { - const mid = (lo + hi) / 2; - const { lines, heightIn } = layoutGivenWidth(paragraphs, mid); - const totalH = heightIn + padding; - if (totalH <= innerH + 1e-6) { - best = mid; - hi = mid; - } else { - lo = mid; - } - } - const { lines, heightIn } = layoutGivenWidth(paragraphs, best); - const contentH = heightIn + padding; - const passthrough = buildPassthroughOptions(opts, fontSizePt, margins); - return { - ...passthrough, - w: best + margins.left + margins.right, - h: contentH + margins.top + margins.bottom, - lines, - contentH, - margins, - topInset: topInsetIn, - }; - } - - throw new Error( - "calcTextBox(): insufficient information. Provide {lines} or ({w,text}) or ({h,text})." - ); -} - -function layoutGivenWidth(paragraphs, boxW) { - let totalLines = 0; - let heightIn = 0; - for (let i = 0; i < paragraphs.length; i++) { - const para = paragraphs[i]; - const widthScale = getWidthScaleForParagraph(para); - const usableWidth = Math.max(0.01, boxW - para.textIndentIn) * widthScale; - const lines = greedyWrap(para, usableWidth); - const count = Math.max(1, lines.length); - totalLines += count; - const lineHeightIn = (para.fontSizePt / 72) * para.leading; - heightIn += count * lineHeightIn; - if (i !== paragraphs.length - 1) heightIn += para.paraSpaceAfterIn; - } - return { lines: totalLines, heightIn }; -} - -function greedyWrap(paragraph, maxWidthIn) { - const text = paragraph.text || ""; - if (text.length === 0) return [""]; - const breaker = new LineBreaker(text); - const breakpoints = []; - let bk; - while ((bk = breaker.nextBreak())) { - breakpoints.push({ pos: bk.position, required: bk.required }); - } - const lines = []; - let start = skipTextWhitespace(text, 0); - let idx = 0; - while (start < text.length) { - while (idx < breakpoints.length && breakpoints[idx].pos <= start) idx++; - let chosen = null; - let probe = idx; - while (probe < breakpoints.length) { - const br = breakpoints[probe]; - const slice = text.slice(start, br.pos); - const width = measureRunWidth(paragraph, trimLineEnd(slice)); - if (width <= maxWidthIn + 1e-6) { - chosen = br; - probe++; - if (br.required) break; - } else { - break; - } - } - if (!chosen) { - const forced = forceBreakSegment(text, start, maxWidthIn, paragraph); - if (forced.segment.length === 0) break; - lines.push(trimLineEnd(forced.segment)); - start = skipTextWhitespace(text, forced.nextIndex); - continue; - } - const lineText = text.slice(start, chosen.pos); - lines.push(trimLineEnd(lineText)); - start = skipTextWhitespace(text, chosen.pos); - } - if (!lines.length) lines.push(""); - return lines; -} - -function splitTextIntoTokens(text) { - if (typeof text !== "string") return [""]; - const tokens = text.split(/(\s+)/); - return tokens.length ? tokens : [""]; -} - -function trimLineEnd(value) { - return typeof value === "string" ? value.replace(/\s+$/u, "") : ""; -} - -function measureRunWidth(paragraph, text) { - if (!text || text.length === 0) return 0; - const fontData = getFontData( - paragraph.fontFace, - paragraph.fontStyle, - paragraph.fontWeight - ); - if (fontData && fontData.font) { - const layout = fontData.font.layout(text); - const widthPts = - (layout.advanceWidth / fontData.font.unitsPerEm) * paragraph.fontSizePt; - return Math.max(0, widthPts / 72); - } - return TEXT_MEASURER( - text, - paragraph.fontSizePt, - paragraph.fontFace, - paragraph.fontStyle, - paragraph.fontWeight - ); -} - -function forceBreakSegment(text, start, maxWidthIn, paragraph) { - const chars = Array.from(text.slice(start)); - if (chars.length === 0) return { segment: "", nextIndex: text.length }; - let buffer = ""; - let consumedUnits = 0; - for (let i = 0; i < chars.length; i++) { - const candidate = buffer + chars[i]; - const width = measureRunWidth(paragraph, trimLineEnd(candidate)); - if (width <= maxWidthIn + 1e-6) { - buffer = candidate; - consumedUnits += chars[i].length; - continue; - } - if (buffer.length === 0) { - buffer = chars[i]; - consumedUnits += chars[i].length; - } - break; - } - if (buffer.length === 0) { - buffer = chars[0] || ""; - consumedUnits = buffer.length; - } - return { segment: buffer, nextIndex: start + consumedUnits }; -} - -function skipTextWhitespace(text, index) { - let idx = index; - while (idx < text.length && /\s/.test(text[idx])) idx++; - return idx; -} - -function buildParagraphModels(textOrRuns, baseStyle) { - const entries = collectParagraphEntries(textOrRuns); - if (entries.length === 0) { - return [resolveParagraphStyle({ text: "" }, baseStyle)]; - } - return entries.map((entry) => resolveParagraphStyle(entry, baseStyle)); -} - -function collectParagraphEntries(textOrRuns) { - const result = []; - if (Array.isArray(textOrRuns)) { - for (const entry of textOrRuns) { - if (typeof entry === "string") { - pushParagraphSegments(entry, undefined, result); - } else if (entry && typeof entry === "object") { - pushParagraphSegments(entry.text ?? "", entry.options || {}, result); - } - } - return result; - } - pushParagraphSegments(textOrRuns ?? "", undefined, result); - return result; -} - -function pushParagraphSegments(text, options, target) { - const normalized = String(text ?? ""); - const parts = normalized.split(/\r?\n/); - if (parts.length === 0) { - target.push({ text: "", options }); - return; - } - for (const part of parts) { - target.push({ text: part, options }); - } -} - -function resolveParagraphStyle(entry, baseStyle) { - const opts = entry.options || {}; - const fontFace = - (opts.fontFace && String(opts.fontFace).trim()) || - baseStyle.fontFace || - "Arial"; - const fontStyle = - opts.italic === true || opts.fontStyle === "italic" - ? "italic" - : baseStyle.fontStyle || "normal"; - const fontWeight = - opts.bold === true || String(opts.fontWeight || "").toLowerCase() === "bold" - ? "bold" - : baseStyle.fontWeight || "normal"; - const fontSizePt = - toNumber(opts.fontSize, baseStyle.fontSizePt) || baseStyle.fontSizePt; - const leading = - toNumber(opts.leading, baseStyle.leading) || baseStyle.leading || 1.15; - const paraSpaceAfterPt = - toNumber(opts.paraSpaceAfter, baseStyle.paraSpaceAfterPt) || - baseStyle.paraSpaceAfterPt || - 0; - const hasBullet = !!opts.bullet; - let indentPt = toNumber(opts.indent, NaN); - if (!Number.isFinite(indentPt) && hasBullet) { - indentPt = toNumber(opts.bullet.indent, NaN); - } - if (!Number.isFinite(indentPt)) indentPt = 0; - const hangingPt = toNumber(opts.hanging, 0) || 0; - let textIndentIn = 0; - if (indentPt > 0) { - if (hasBullet) { - // PowerPoint-style bullets: "indent" is the distance from the left edge - // of the text box to the start of the text (the bullet itself is hung - // using the hanging value). This means the available width for the text - // is boxWidth - indent, not boxWidth - (indent - hanging). Modeling it - // this way matches the manual line counts from PowerPoint/LibreOffice. - textIndentIn = indentPt / 72; - } else { - // Non-bullet paragraphs keep the prior behavior where hanging reduces - // the effective indent (similar to CSS text-indent). - textIndentIn = Math.max(0, (indentPt - hangingPt) / 72); - } - } - return { - text: entry.text || "", - fontFace, - fontStyle, - fontWeight, - fontSizePt, - leading, - paraSpaceAfterIn: paraSpaceAfterPt / 72, - textIndentIn, - }; -} - -function getFontData(face, fontStyle, fontWeight) { - const key = makeFontCacheKey(face, fontStyle, fontWeight); - if (fontKitCache.has(key)) return fontKitCache.get(key); - const fontPath = findFontPath(face, fontStyle, fontWeight); - if (!fontPath) { - fontKitCache.set(key, null); - return null; - } - try { - let font = fontkit.openSync(fontPath); - if (font && typeof font.fonts === "object") { - font = selectCollectionFont(font, fontStyle, fontWeight); - } - if (!font || typeof font.layout !== "function") { - fontKitCache.set(key, null); - return null; - } - registerCanvasFontVariant(fontPath, face, fontStyle, fontWeight, key); - const payload = { font, path: fontPath }; - fontKitCache.set(key, payload); - return payload; - } catch (err) { - fontKitCache.set(key, null); - return null; - } -} - -function makeFontCacheKey(face, fontStyle, fontWeight) { - const family = (face || "Arial").trim(); - const style = (fontStyle || "normal").toLowerCase(); - const weight = (fontWeight || "normal").toLowerCase(); - return `${family}::${style}::${weight}`; -} - -function registerCanvasFontVariant( - fontPath, - face, - fontStyle, - fontWeight, - cacheKey -) { - if (registeredFontVariants.has(cacheKey)) return; - try { - Canvas.registerFont(fontPath, { - family: face, - style: fontStyle || "normal", - weight: fontWeight || "normal", - }); - registeredFontVariants.add(cacheKey); - } catch (err) { - // ignore registration failure; measurement will fall back to Skia default - } -} - -function findFontPath(face, fontStyle, fontWeight) { - const family = (face || "").trim(); - if (family.length === 0) return null; - const key = makeFontCacheKey(family, fontStyle, fontWeight); - if (fontPathCache.has(key)) return fontPathCache.get(key); - const styleParts = []; - if ((fontWeight || "").toLowerCase() === "bold") styleParts.push("Bold"); - if ((fontStyle || "").toLowerCase() === "italic") styleParts.push("Italic"); - const styleQuery = - styleParts.length > 0 ? `:style=${styleParts.join(" ")}` : ""; - const query = `${family}${styleQuery}`; - const result = spawnSync("fc-match", ["-f", "%{file}", query], { - encoding: "utf8", - }); - if (result.status === 0) { - const output = String(result.stdout || "").trim(); - if (output.length > 0) { - fontPathCache.set(key, output); - return output; - } - } - fontPathCache.set(key, null); - return null; -} - -function selectCollectionFont(collection, fontStyle, fontWeight) { - const fonts = collection.fonts || []; - if (fonts.length === 0) return null; - const wantItalic = (fontStyle || "").toLowerCase() === "italic"; - const wantBold = (fontWeight || "").toLowerCase() === "bold"; - let best = fonts[0]; - let bestScore = scoreFontVariant(best, wantItalic, wantBold); - for (let i = 1; i < fonts.length; i++) { - const candidate = fonts[i]; - const score = scoreFontVariant(candidate, wantItalic, wantBold); - if (score > bestScore) { - best = candidate; - bestScore = score; - } - } - return best; -} - -function scoreFontVariant(font, wantItalic, wantBold) { - if (!font) return -1; - const name = String(font.fullName || font.postscriptName || "").toLowerCase(); - const isItalic = /italic|oblique/.test(name); - const isBold = /bold|black|heavy|semibold|extrabold/.test(name); - let score = 0; - if (isItalic === wantItalic) score += 1; - if (isBold === wantBold) score += 1; - return score; -} - -// Empirical width scaling to better match PowerPoint/LibreOffice line breaks. -// A tiny global shrink (about -1.5%) nudges borderline words to wrap the same -// way Office does, with per-script tweaks for cases where our measurer -// systematically under- or over-estimates glyph widths. We intentionally avoid -// per-font calibration so this helper generalizes beyond the regression deck. -function getWidthScaleForParagraph(paragraph) { - if (!paragraph || typeof paragraph.text !== "string") return 1; - const text = paragraph.text; - // Thai script: our measurer tends to slightly over-estimate, which can cause - // extra wraps. Give it a bit more room horizontally. - if (/[ก-๛]/u.test(text)) { - return 1.2; - } - - // Arabic: we usually underestimate, so shrink available width a bit more to - // encourage earlier breaks. - if (/[\u0600-\u06FF]/u.test(text)) { - return 0.97; - } - - // Base shrink for most Latin and other scripts. - return 0.985; -} - -// Build options to pass directly to pptx.addText. We exclude measurement-only -// fields and fill sensible defaults (e.g., fontSize) so callers can spread -// the result into addText just like the image sizing helpers. -function buildPassthroughOptions(opts, fontSizePt, margins) { - const exclude = new Set([ - "text", - "lines", - "w", // will be set by calcTextBox - "h", // will be set by calcTextBox - // fontFace/style/weight are useful for addText; allow passthrough - "leading", - "padding", - ]); - const out = {}; - for (const k of Object.keys(opts)) { - if (!exclude.has(k)) out[k] = opts[k]; - } - if (out.fontSize == null) out.fontSize = fontSizePt; - if (opts.margin != null) out.margin = margins; - return out; -} - -function getTextMeasurer() { - // Skia-canvas only for accurate shaping and Fontconfig-based resolution. - // Throws if skia-canvas is not available. - const canvas = new Canvas(2, 2); - const ctx = canvas.getContext("2d"); - const PX_PER_IN = 96; - return (text, fontSizePt, fontFace, fontStyle, fontWeight) => { - const px = (fontSizePt / 72) * PX_PER_IN; - const style = fontStyle || "normal"; - const weight = fontWeight || "normal"; - // CSS shorthand: style weight size family - ctx.font = `${style} ${weight} ${px}px ${fontFace || "Arial"}`; - const metrics = ctx.measureText(text); - return (metrics.width || 0) / PX_PER_IN; - }; -} - -function normalizeMargins(m) { - const toInches = (value) => - typeof value === "number" && Number.isFinite(value) ? value / 72 : 0; - if (m && typeof m === "object") { - if (Number.isFinite(m.left) || Number.isFinite(m.top)) { - return { - left: toInches(m.left), - right: toInches(m.right), - top: toInches(m.top), - bottom: toInches(m.bottom), - }; - } - } - const all = toInches(m); - return { left: all, right: all, top: all, bottom: all }; -} - -function normalizeText(textOrRuns) { - if (Array.isArray(textOrRuns)) { - return textOrRuns - .map((item) => { - if (typeof item === "string") return item; - if (item && typeof item.text === "string") return item.text; - return ""; - }) - .join(""); - } - return typeof textOrRuns === "string" ? textOrRuns : String(textOrRuns ?? ""); -} - -function toNumber(v, fallback) { - const n = typeof v === "string" ? parseFloat(v) : v; - return Number.isFinite(n) ? n : fallback; -} - -module.exports = { - calcTextBoxHeightSimple, - calcTextBox, - autoFontSize, -}; diff --git a/.github/skills/openai-slides/assets/pptxgenjs_helpers/util.js b/.github/skills/openai-slides/assets/pptxgenjs_helpers/util.js deleted file mode 100644 index 8a6e1fc2..00000000 --- a/.github/skills/openai-slides/assets/pptxgenjs_helpers/util.js +++ /dev/null @@ -1,24 +0,0 @@ -// Copyright (c) OpenAI. All rights reserved. -"use strict"; - -// Safe outer shadow helper (avoid inner/outer mix and XML pitfalls) -function safeOuterShadow( - color = "000000", - opacity = 0.25, - angle = 45, - blur = 3, - offset = 2 -) { - return { - type: "outer", - color, - opacity, - angle, - blur, - offset, - }; -} - -module.exports = { - safeOuterShadow, -}; diff --git a/.github/skills/openai-slides/assets/slides-small.svg b/.github/skills/openai-slides/assets/slides-small.svg deleted file mode 100644 index 8afd52d9..00000000 --- a/.github/skills/openai-slides/assets/slides-small.svg +++ /dev/null @@ -1,3 +0,0 @@ - - - diff --git a/.github/skills/openai-slides/assets/slides.png b/.github/skills/openai-slides/assets/slides.png deleted file mode 100644 index c05e3091..00000000 Binary files a/.github/skills/openai-slides/assets/slides.png and /dev/null differ diff --git a/.github/skills/openai-slides/references/pptxgenjs-helpers.md b/.github/skills/openai-slides/references/pptxgenjs-helpers.md deleted file mode 100644 index 1a564deb..00000000 --- a/.github/skills/openai-slides/references/pptxgenjs-helpers.md +++ /dev/null @@ -1,61 +0,0 @@ -# PptxGenJS Helpers - -## When To Read This - -Read this file when you need helper API details, command examples for the bundled Python scripts, or dependency notes for a slide-generation task. - -## Helper Modules - -- `autoFontSize(textOrRuns, fontFace, opts)`: Pick a font size that fits a fixed box. -- `calcTextBox(fontSizePt, opts)`: Estimate text-box geometry from font size and content. -- `calcTextBoxHeightSimple(fontSizePt, numLines, leading?, padding?)`: Quick text height estimate. -- `imageSizingCrop(pathOrData, x, y, w, h)`: Center-crop an image into a target box. -- `imageSizingContain(pathOrData, x, y, w, h)`: Fit an image fully inside a target box. -- `svgToDataUri(svgString)`: Convert an SVG string into an embeddable data URI. -- `latexToSvgDataUri(texString)`: Render LaTeX to SVG for crisp equations. -- `getImageDimensions(pathOrData)`: Read image width, height, type, and aspect ratio. -- `safeOuterShadow(...)`: Build a safe outer-shadow config for PowerPoint output. -- `codeToRuns(source, language)`: Convert source code into rich-text runs for `addText`. -- `warnIfSlideHasOverlaps(slide, pptx)`: Emit overlap warnings for diagnostics. -- `warnIfSlideElementsOutOfBounds(slide, pptx)`: Emit boundary warnings for diagnostics. -- `alignSlideElements(slide, indices, alignment)`: Align selected elements precisely. -- `distributeSlideElements(slide, indices, direction)`: Evenly space selected elements. - -## Dependency Notes - -JavaScript helpers expect these packages when you use the corresponding features: - -- Core authoring: `pptxgenjs` -- Text measurement: `skia-canvas`, `linebreak`, `fontkit` -- Syntax highlighting: `prismjs` -- LaTeX rendering: `mathjax-full` - -Python scripts expect these packages: - -- `Pillow` -- `pdf2image` -- `python-pptx` -- `numpy` - -System tools used by the Python scripts: - -- `soffice` / LibreOffice for PPTX to PDF conversion -- Poppler tools for PDF size/raster support used by `pdf2image` -- `fc-list` for font inspection -- Optional rasterization tools for `ensure_raster_image.py`: Inkscape, ImageMagick, Ghostscript, `heif-convert`, `JxrDecApp` - -## Script Notes - -- `render_slides.py`: Convert a deck to PNGs. Good for visual review and diffing. -- `slides_test.py`: Add a gray border outside the original canvas, render, and check whether any content leaks into the border. -- `create_montage.py`: Combine multiple rendered slide images into a single overview image. -- `detect_font.py`: Distinguish between fonts that are missing entirely and fonts that are installed but substituted during rendering. -- `ensure_raster_image.py`: Produce a PNG from common vector or unusual raster formats so you can inspect or place the asset easily. - -## Practical Rules - -- Default to `LAYOUT_WIDE` unless the source material says otherwise. -- Set font families explicitly before measuring text. -- Use `valign: "top"` for content boxes that may grow. -- Prefer native PowerPoint charts over rendered images when the chart is simple and likely to be edited later. -- Use SVG instead of PNG for diagrams whenever possible. diff --git a/.github/skills/openai-slides/scripts/create_montage.py b/.github/skills/openai-slides/scripts/create_montage.py deleted file mode 100644 index 8c385c74..00000000 --- a/.github/skills/openai-slides/scripts/create_montage.py +++ /dev/null @@ -1,300 +0,0 @@ -#!/usr/bin/env python3 -# Copyright (c) OpenAI. All rights reserved. -import argparse -import re -import sys -import tempfile -from math import ceil -from os import listdir -from os.path import basename, expanduser, isfile, join, splitext -from pathlib import Path -from typing import Literal - -SCRIPT_DIR = Path(__file__).resolve().parent -if str(SCRIPT_DIR) not in sys.path: - sys.path.insert(0, str(SCRIPT_DIR)) - -from ensure_raster_image import SUPPORTED_EXTS, ensure_raster_image # type: ignore -from PIL import Image, ImageDraw, ImageFont, ImageOps - - -def _make_placeholder(w: int, h: int) -> Image.Image: - """Create a visible placeholder tile with a light gray fill and a red X cross.""" - ph = Image.new("RGBA", (w, h), (220, 220, 220, 255)) - ph_draw = ImageDraw.Draw(ph) - line_color = (180, 0, 0, 255) - ph_draw.line([(0, 0), (ph.width - 1, ph.height - 1)], fill=line_color, width=3) - ph_draw.line([(ph.width - 1, 0), (0, ph.height - 1)], fill=line_color, width=3) - return ph - - -def _load_images_with_placeholders( - input_files: list[str], retain_converted_files: bool, fail_on_image_error: bool = False -) -> tuple[list[str], list[Image.Image | None]]: - labels = [basename(p) for p in input_files] - images: list[Image.Image | None] = [] - if retain_converted_files: - for p in input_files: - try: - images.append(Image.open(ensure_raster_image(p))) - except Exception as e: - if fail_on_image_error: - raise - print(f'Warning: Failed to convert or load image "{p}": {e}') - images.append(None) - else: - with tempfile.TemporaryDirectory(prefix="montage_convert_") as tmp_conv: - for p in input_files: - try: - images.append(Image.open(ensure_raster_image(p, tmp_conv))) - except Exception as e: - if fail_on_image_error: - raise - print(f'Warning: Failed to convert or load image "{p}": {e}') - images.append(None) - return labels, images - - -def _natural_key(s: str) -> list: - """Key function for natural sorting (e.g., Slide2 before Slide10).""" - return [int(part) if part.isdigit() else part for part in re.split(r"(\d+)", s)] - - -def create_montage( - input_files: list[str], - output_file: str, - num_col: int, - cell_w: int, - cell_h: int, - gap: int, - label_mode: Literal["number", "filename", "none"], - retain_converted_files: bool = False, - fail_on_image_error: bool = False, -) -> None: - """Build a montage with a fixed number of columns. - - Each cell has size `cell_w` x `cell_h`. Every input image is resized isotropically to fit inside - the cell. `gap` controls spacing around and between cells (outer margin equals gap). - Label behavior is controlled by `label_mode` which can be one of: - - "none": no labels are drawn - - "number": draw a 1-based index beneath each image - - "filename": draw the filename (no directory) beneath each image - """ - - if num_col <= 0: - raise ValueError("num_col must be positive") - if cell_w <= 0 or cell_h <= 0: - raise ValueError("cell_w and cell_h must be positive") - - labels, images = _load_images_with_placeholders( - input_files=input_files, - retain_converted_files=retain_converted_files, - fail_on_image_error=fail_on_image_error, - ) - - num_images = len(images) - num_valid = sum(1 for im in images if im is not None) - if num_valid == 0: - raise ValueError("No valid images to render.") - if num_valid < num_images: - cell_size = round(min(cell_w, cell_h) * 0.6) - placeholder = _make_placeholder(cell_size, cell_size) - else: - placeholder = None - cols = num_col - rows = ceil(num_images / cols) - - temp_canvas = Image.new("RGB", (10, 10), (255, 255, 255)) - temp_draw = ImageDraw.Draw(temp_canvas) - - # Choose a readable default font size relative to cell height - font: ImageFont.FreeTypeFont | ImageFont.ImageFont - try: - # Attempt to use a common system font for clarity; fallback to default - font_size = max(12, min(36, int(cell_h * 0.12))) - font = ImageFont.truetype("arial.ttf", font_size) - except Exception: - font = ImageFont.load_default() - # Adjust default font effect size estimate - font_size = 12 - - draw_labels = label_mode != "none" - label_height = 0 - if draw_labels: - # Height is approximately constant across strings for a given font - # Use 'Ag' to approximate ascent ('A') and descender ('g') for filename text - sample_text = "1" if label_mode == "number" else "Ag" - lbbox = temp_draw.textbbox((0, 0), sample_text, font=font) - label_height = ceil(lbbox[3] - lbbox[1]) + 6 - - row_h = cell_h + label_height - - canvas_w = cols * cell_w + (cols + 1) * gap - canvas_h = rows * row_h + (rows + 1) * gap - # Light grey canvas background as in typical slide sorter view - canvas = Image.new("RGB", (canvas_w, canvas_h), (242, 242, 242)) - draw = ImageDraw.Draw(canvas) - - for idx, img in enumerate(images): - col = idx % cols - row = idx // cols - - # Top-left corner of the cell including outer margin and gaps - x0 = gap + col * (cell_w + gap) - y0 = gap + row * (row_h + gap) - - # Fit the image within the cell while preserving aspect ratio - if label_mode == "number": - label = str(idx + 1) - elif label_mode == "filename": - label = labels[idx] - else: - label = "" - - if draw_labels: - bbox = draw.textbbox((0, 0), label, font=font) - text_w = bbox[2] - bbox[0] - else: - text_w = 0 - - if img: - resized = ImageOps.contain( - img.convert("RGBA"), - (cell_w, cell_h), - method=Image.Resampling.LANCZOS, - ) - else: - print(f"Warning: Using placeholder for invalid image at row={row + 1}, col={col + 1}") - assert placeholder is not None - resized = placeholder - - paste_x = x0 + (cell_w - resized.width) // 2 - paste_y = y0 + (cell_h - resized.height) // 2 - canvas.paste( - resized, - (paste_x, paste_y), - mask=resized.split()[3] if resized.mode == "RGBA" else None, - ) - - border_color = (160, 160, 160) - bw = 1 - draw.rectangle( - [ - paste_x - bw, - paste_y - bw, - paste_x + resized.width, - paste_y + resized.height, - ], - outline=border_color, - width=bw, - ) - - if draw_labels: - tx = x0 + round((cell_w - text_w) / 2) - ty = y0 + cell_h + 3 - draw.text((tx, ty), label, font=font, fill=(0, 0, 0)) - - canvas.save(output_file) - print(f"Montage saved to {output_file}") - - -def main() -> None: - parser = argparse.ArgumentParser( - description=( - "Create a montage with a fixed number of columns. " - "Each image is resized isotropically to fit inside a cell of size (cell_width x cell_height)." - ) - ) - group = parser.add_mutually_exclusive_group(required=True) - group.add_argument("--input_files", nargs="+", help="List of input image file paths") - group.add_argument("--input_dir", help="Directory containing input images") - parser.add_argument( - "--output_file", - required=True, - help=( - "Path to save the output montage image. The format is inferred from the file extension." - ), - ) - parser.add_argument( - "--num_col", - type=int, - default=5, - help="Number of images per row (default: 5)", - ) - parser.add_argument( - "--cell_width", - type=int, - default=400, - help="Container width in pixels for each image (default: 400)", - ) - parser.add_argument( - "--cell_height", - type=int, - default=225, - help="Container height in pixels for each image (default: 225)", - ) - parser.add_argument( - "--gap", - type=int, - default=16, - help="Gap in pixels between images and canvas margins (default: 16)", - ) - parser.add_argument( - "--label_mode", - choices=["number", "filename", "none"], - default="number", - help=( - "Label mode: 'number' to draw 1-based indices (default), 'filename' to use the " - "image's filename (no directory), or 'none' for no labels" - ), - ) - parser.add_argument( - "--retain_converted_files", - action="store_true", - default=False, - help=( - "If set, write converted images (e.g., SVG->PNG, WDP->PNG) next to the original files " - "instead of a temporary directory." - ), - ) - parser.add_argument( - "--fail_on_image_error", - action="store_true", - default=False, - help=( - "If set, fail immediately when any image conversion/loading fails (no placeholders). " - "By default, failures are tolerated and placeholders are used." - ), - ) - args = parser.parse_args() - - output_path = expanduser(args.output_file) - if args.input_files: - input_files = [expanduser(p) for p in args.input_files] - else: - input_dir = expanduser(args.input_dir) - names = sorted(listdir(input_dir), key=_natural_key) - dir_entries = [join(input_dir, f) for f in names] - input_files = [ - p for p in dir_entries if isfile(p) and splitext(p)[1].lower() in SUPPORTED_EXTS - ] - if not input_files: - raise ValueError( - "No image files with supported extensions were found in the specified directory." - ) - - create_montage( - input_files=input_files, - output_file=output_path, - num_col=args.num_col, - cell_w=args.cell_width, - cell_h=args.cell_height, - gap=args.gap, - label_mode=args.label_mode, - retain_converted_files=args.retain_converted_files, - fail_on_image_error=args.fail_on_image_error, - ) - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-slides/scripts/detect_font.py b/.github/skills/openai-slides/scripts/detect_font.py deleted file mode 100644 index a5fc3937..00000000 --- a/.github/skills/openai-slides/scripts/detect_font.py +++ /dev/null @@ -1,873 +0,0 @@ -#!/usr/bin/env python3 -"""Copyright (c) OpenAI. All rights reserved. - -Detect missing fonts for PPTX rendering by converting to ODP and inspecting the resolved font -families per slide. - -Overview -======== -PowerPoint files (PPTX) declare requested font families in runs and theme defaults, but the actual -font used at render time depends on the renderer (LibreOffice in our pipeline), platform -availability, and style inheritance. To make detection stable and renderer-accurate, this module: - -- Extracts requested families from PPTX per slide (reads a:r/a:rPr plus document defaults, grouped - by script: latin/ea/cs/sym). Analysis is done per run: we infer the script from run text and - select the matching a:rPr child (e.g., latin/ea/cs). Fonts declared for other scripts in the same - run are not counted as used. -- Converts the PPTX to ODP using headless LibreOffice and parses ODP content.xml and styles.xml to - discover which families LibreOffice actually resolved for each slide (including master pages and - defaults). -- Classifies each requested family on each slide into two buckets: - - font_missing: the family is not installed on the system (per fontconfig synonyms), so resolution - cannot possibly match the request. - - font_substituted: the family is installed but was resolved to another family in ODP for the - slide (theme/style inheritance or glyph coverage), i.e., installed but substituted. - -Key Design ------------------------ -1) Inspect the renderer's decision, not only the author's request. Reading PPTX alone tells you what - was requested, not what LibreOffice will choose after applying styles and availability checks. - Converting to ODP and reading the resolved fo:font-family/style:font-name* values yields a - faithful view of what the renderer actually used for each slide. - -2) Robust style resolution across ODP structures. Fonts can be specified under multiple layers. We - parse office:automatic-styles (both content.xml and styles.xml), office:styles and - style:default-style, draw:master-page references used by slides, nested style:text-properties - under paragraph-properties, and parent style chains (style:parent-style-name). A text-based - fallback parser supplements XML namespace lookups when vendor XML variations occur. - -3) Scalable aliasing via fontconfig synonyms, not ad hoc maps. PostScript names, full names, and - family names often differ. We build a synonym map from fc-list that unifies those identifiers. We - deliberately do NOT use fc-match -s fallback chains for matching, because fallback families - (e.g., DejaVu Sans) would mask missing/substitution cases and produce false passes. - -4) Clear classification: missing vs substituted. - - Missing: no synonym of the requested base family is present in the installed font set (per - fontconfig). These require installation. - - Substituted: the family is installed, but ODP does not reference it on the slide (LibreOffice - chose another family), which is useful for diagnosing style/theme issues or glyph-coverage - driven substitutions. - -Not Chosen (and why) --------------------- -- PDF inspection (e.g., pdffonts): PostScript names don't reliably map back to authoring families; - PDFs often reflect subsetted fonts and fallback choices, making robust detection noisy. -- Ad hoc alias tables: unscalable for large-scale fonts and platform variants; the fontconfig - synonym corpus covers family/fullname/PostScript consistently. -- Treating fallback families as matches (fc-match -s): causes false negatives by accepting generic - fallbacks when the requested family is missing. -- Hardcoding checks in the renderer: we keep detection separate from render_slides to avoid - coupling and allow standalone checking. - -CLI ---- -- JSON output exposes two categories by default (and text mode mirrors them): font_missing_overall/ - font_missing_by_slide and font_substituted_overall/font_substituted_by_slide. -- Flags include_missing/include_substituted control which categories are emitted (default True/True). -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -import tempfile -import xml.etree.ElementTree as ET -from functools import lru_cache -from os.path import abspath, basename, exists, expanduser, join, splitext -from zipfile import ZipFile - -STYLE_TOKENS = [ - "regular", - "condensed", - "compressed", - "narrow", - "italic", - "oblique", - "semibold", - "demibold", - "bold", - "black", - "extra light", - "ultra light", - "extralight", - "ultralight", - "light", - "thin", - "medium", -] - - -def normalize_font_family_name(name: str) -> str: - s = name.casefold() - s = re.sub(r"\([^)]*\)", " ", s) - s = re.sub(r"[\s\-\_\.,/\'\"]+", " ", s) - return s.strip() - - -def _or_dummy(node: ET.Element | None) -> ET.Element: - """Return the element if not None, otherwise a harmless dummy element. - - Avoids deprecated truthiness checks on Element instances (`elem or dummy`). - """ - return node if node is not None else ET.Element("dummy") - - -@lru_cache(maxsize=1) -def _build_fc_synonym_map() -> dict[str, set[str]]: - """Build synonym map from fontconfig; raise on failures; memoized (size=1).""" - proc = subprocess.run( - [ - "fc-list", - "--format", - "%{family}\t%{fullname}\t%{postscriptname}\n", - ], - capture_output=True, - text=True, - check=True, - ) - syn: dict[str, set[str]] = {} - for line in (proc.stdout or "").splitlines(): - parts = line.split("\t") - if len(parts) != 3: - continue - fam_field, full_field, ps_field = parts - names: set[str] = set() - for field in (fam_field, full_field, ps_field): - for item in field.split(","): - norm = normalize_font_family_name(item) - if norm: - names.add(norm) - names.add(norm.replace(" ", "")) - for name in list(names): - bucket = syn.setdefault(name, set()) - bucket.update(names) - return syn - - -def _expand_via_fontconfig(family_base_norm: str) -> set[str]: - # Accept only true aliases/synonyms (family/fullname/PostScript) — not fallback replacements - acceptable: set[str] = {family_base_norm, family_base_norm.replace(" ", "")} - syn = _build_fc_synonym_map() - if family_base_norm in syn: - acceptable.update(syn[family_base_norm]) - no_space = family_base_norm.replace(" ", "") - if no_space in syn: - acceptable.update(syn[no_space]) - return acceptable - - -def parse_font_family_base_and_styles(name_norm: str) -> tuple[str, set[str]]: - tokens = name_norm.split() - required: set[str] = set() - weight_code_map = { - "25": "ultra light", - "35": "thin", - "45": "light", - "55": "regular", - "65": "medium", - "75": "bold", - "85": "black", - "95": "black", - } - if tokens and tokens[0].isdigit() and tokens[0] in weight_code_map: - required.add(weight_code_map[tokens[0]]) - tokens = tokens[1:] - if len(tokens) == 1: - t = tokens[0] - fused_map = [ - ("extralight", "extra light"), - ("ultralight", "ultra light"), - ("semibold", "semibold"), - ("demibold", "semibold"), - ("condensed", "condensed"), - ("compressed", "condensed"), - ("narrow", "condensed"), - ("italic", "italic"), - ("oblique", "italic"), - ("bold", "bold"), - ("black", "black"), - ("light", "light"), - ("thin", "thin"), - ("medium", "medium"), - ("regular", "regular"), - ] - changed = True - while changed: - changed = False - for suf, tok in fused_map: - if t.endswith(suf) and len(t) > len(suf): - t = t[: -len(suf)] - required.add(tok) - changed = True - break - return (t.strip(), required) - - while tokens: - tail = " ".join(tokens[-2:]) if len(tokens) >= 2 else tokens[-1] - matched = None - for style in STYLE_TOKENS: - if tail == style: - matched = style - break - if matched is None and tokens[-1] in STYLE_TOKENS: - matched = tokens[-1] - if matched is None: - break - if matched in ("compressed", "narrow"): - required.add("condensed") - elif matched == "roman": - required.add("regular") - elif matched == "demibold": - required.add("semibold") - else: - required.add(matched) - if " " in matched: - tokens = tokens[:-2] - else: - tokens = tokens[:-1] - return (" ".join(tokens).strip(), required) - - -def _split_odf_family_list(value: str) -> list[str]: - out: list[str] = [] - for part in value.split(","): - p = part.strip().strip("\"' ") - if p: - out.append(normalize_font_family_name(p)) - return out - - -def extract_used_fonts_from_pptx(pptx_path: str) -> dict[int, set[str]]: - by_slide: dict[int, set[str]] = {} - with ZipFile(pptx_path, "r") as zf: - for name in zf.namelist(): - if not (name.startswith("ppt/slides/slide") and name.endswith(".xml")): - continue - base = os.path.basename(name) - m = re.search(r"(?i)slide(\d+)\.xml$", base) - slide_num = int(m.group(1)) if m else None - with zf.open(name) as f: - tree = ET.parse(f) - root = tree.getroot() - ns = {"a": "http://schemas.openxmlformats.org/drawingml/2006/main"} - defaults = _collect_default_font_faces(root) - for r in root.findall(".//a:r", ns): - parts: list[str] = [] - for t in r.findall("a:t", ns): - if t.text: - parts.append(t.text) - text = "".join(parts) - if not text: - continue - script = _detect_script_tag(text) - rpr = r.find("a:rPr", ns) - face_norm: str | None = None - if rpr is not None: - child = rpr.find(f"a:{script}", ns) - if child is not None: - face = child.get("typeface") - if face and not face.startswith("+"): - face_norm = normalize_font_family_name(face) - bucket = by_slide.setdefault(slide_num or -1, set()) - if face_norm is None: - for f in defaults.get(script, set()): - bucket.add(f) - else: - bucket.add(face_norm) - return {k: v for k, v in by_slide.items() if k is not None and k != -1} - - -def _detect_script_tag(text: str) -> str: - for ch in text: - cp = ord(ch) - if ( - 0x4E00 <= cp <= 0x9FFF - or 0x3400 <= cp <= 0x4DBF - or 0xF900 <= cp <= 0xFAFF - or 0x3040 <= cp <= 0x309F - or 0x30A0 <= cp <= 0x30FF - or 0x31F0 <= cp <= 0x31FF - or 0xAC00 <= cp <= 0xD7AF - or 0x3100 <= cp <= 0x312F - or 0x3000 <= cp <= 0x303F - ): - return "ea" - for ch in text: - cp = ord(ch) - if ( - 0x0590 <= cp <= 0x05FF - or 0x0600 <= cp <= 0x06FF - or 0x0700 <= cp <= 0x077F - or 0x0780 <= cp <= 0x07BF - or 0x0900 <= cp <= 0x0D7F - or 0x0E00 <= cp <= 0x0E7F - or 0x0E80 <= cp <= 0x0EFF - or 0xFB50 <= cp <= 0xFDFF - or 0xFE70 <= cp <= 0xFEFF - ): - return "cs" - for ch in text: - cp = ord(ch) - if ( - (0x0041 <= cp <= 0x005A) - or (0x0061 <= cp <= 0x007A) - or (0x0030 <= cp <= 0x0039) - or (0x00C0 <= cp <= 0x024F) - or (0x1E00 <= cp <= 0x1EFF) - ): - return "latin" - return "latin" - - -def _collect_default_font_faces(root: ET.Element) -> dict[str, set[str]]: - ns = {"a": "http://schemas.openxmlformats.org/drawingml/2006/main"} - defaults: dict[str, set[str]] = {"latin": set(), "ea": set(), "cs": set(), "sym": set()} - for defrpr in root.findall(".//a:defRPr", ns): - for tag in ("latin", "ea", "cs", "sym"): - child = defrpr.find(f"a:{tag}", ns) - if child is not None: - face = child.get("typeface") - if face and not face.startswith("+"): - defaults[tag].add(normalize_font_family_name(face)) - return defaults - - -def _run_soffice_convert(cmd: list[str]) -> None: - subprocess.run( - cmd, - check=False, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - env=os.environ.copy(), - ) - - -def _export_to_odp(pptx_path: str, user_profile: str, out_dir: str, stem: str) -> str: - bin_path = shutil.which("soffice") or shutil.which("libreoffice") or "/usr/bin/libreoffice" - cmd_odp = [ - bin_path, - "-env:UserInstallation=file://" + user_profile, - "--invisible", - "--headless", - "--norestore", - "--convert-to", - "odp", - "--outdir", - out_dir, - pptx_path, - ] - _run_soffice_convert(cmd_odp) - odp_path = join(out_dir, f"{stem}.odp") - return odp_path if exists(odp_path) else "" - - -def _collect_face_map(root: ET.Element, ns: dict[str, str]) -> dict[str, str]: - face_map: dict[str, str] = {} - decls = root.find("office:font-face-decls", ns) - if decls is None: - return face_map - for ff in decls.findall("style:font-face", ns): - name_attr = ff.get("{urn:oasis:names:tc:opendocument:xmlns:style:1.0}name") or ff.get( - "style:name" - ) - fam_attr = ff.get("{urn:oasis:names:tc:opendocument:xmlns:svg-compatible:1.0}font-family") - if not name_attr or not fam_attr: - continue - face_map[normalize_font_family_name(name_attr)] = normalize_font_family_name(fam_attr) - return face_map - - -def _families_from_text_properties( - tp: ET.Element, ns: dict[str, str], face_map: dict[str, str] -) -> set[str]: - fams: set[str] = set() - # Inspect current node for direct font-family - fam_attr = tp.get("{urn:oasis:names:tc:opendocument:xmlns:xsl-fo-compatible:1.0}font-family") - if fam_attr: - fams.update(_split_odf_family_list(fam_attr)) - # Inspect font-name aliases on current node - for key in ( - "{urn:oasis:names:tc:opendocument:xmlns:style:1.0}font-name", - "style:font-name", - "style:font-name-asian", - "style:font-name-complex", - ): - val = tp.get(key) - if val: - norm_val = normalize_font_family_name(val) - mapped = face_map.get(norm_val) - if mapped: - fams.add(normalize_font_family_name(mapped)) - else: - fams.add(norm_val) - # Some styles nest text-properties under paragraph-properties or default-style blocks - if not fams: - nested = None - # paragraph-properties/text-properties - pp = tp.find("style:paragraph-properties", ns) - if pp is not None: - nested = pp.find("style:text-properties", ns) - if nested is None: - # When tp is actually the style:style node, try finding child text-properties directly - nested = tp.find("style:text-properties", ns) - if nested is not None and nested is not tp: - fams.update(_families_from_text_properties(nested, ns, face_map)) - return fams - - -def _extract_styles_from_container( - container: ET.Element | None, ns: dict[str, str], face_map: dict[str, str] -) -> tuple[dict[str, set[str]], set[str]]: - styles: dict[str, set[str]] = {} - defaults: set[str] = set() - if container is None: - return styles, defaults - for st in container.findall("style:style", ns): - name = st.get("{urn:oasis:names:tc:opendocument:xmlns:style:1.0}name") or st.get( - "style:name" - ) - if not name: - continue - fams = _families_from_text_properties( - _or_dummy(st.find("style:text-properties", ns)), ns, face_map - ) - if fams: - styles[name] = fams - for ds in container.findall("style:default-style", ns): - defaults.update( - _families_from_text_properties( - _or_dummy(ds.find("style:text-properties", ns)), ns, face_map - ) - ) - return styles, defaults - - -def _build_style_map( - content: ET.Element, - styles_root: ET.Element | None, - ns: dict[str, str], - face_map: dict[str, str], -) -> tuple[dict[str, set[str]], set[str]]: - style_map: dict[str, set[str]] = {} - default_fams: set[str] = set() - auto_styles = content.find("office:automatic-styles", ns) - styles_part, defaults_part = _extract_styles_from_container(auto_styles, ns, face_map) - style_map.update(styles_part) - default_fams.update(defaults_part) - if styles_root is not None: - # Also parse automatic-styles within styles.xml (document-styles) - styles_auto = styles_root.find("office:automatic-styles", ns) - styles_part, defaults_part = _extract_styles_from_container(styles_auto, ns, face_map) - for k, v in styles_part.items(): - if k not in style_map: - style_map[k] = v - default_fams.update(defaults_part) - common_styles = styles_root.find("office:styles", ns) - styles_part, defaults_part = _extract_styles_from_container(common_styles, ns, face_map) - for k, v in styles_part.items(): - if k not in style_map: - style_map[k] = v - default_fams.update(defaults_part) - # top-level default-style under styles_root - for ds in styles_root.findall("style:default-style", ns): - default_fams.update( - _families_from_text_properties( - _or_dummy(ds.find("style:text-properties", ns)), ns, face_map - ) - ) - # Fallback: include any remaining style:style definitions anywhere in styles.xml - for st in styles_root.findall(".//style:style", ns): - name = st.get("{urn:oasis:names:tc:opendocument:xmlns:style:1.0}name") or st.get( - "style:name" - ) - if not name or name in style_map: - continue - fams = _families_from_text_properties( - _or_dummy(st.find("style:text-properties", ns)), ns, face_map - ) - if fams: - style_map[name] = fams - # also check top-level default-style in content root - for ds in content.findall("style:default-style", ns): - default_fams.update( - _families_from_text_properties( - _or_dummy(ds.find("style:text-properties", ns)), ns, face_map - ) - ) - # Fallback: include any remaining style:style definitions anywhere in content.xml - for st in content.findall(".//style:style", ns): - name = st.get("{urn:oasis:names:tc:opendocument:xmlns:style:1.0}name") or st.get( - "style:name" - ) - if not name or name in style_map: - continue - fams = _families_from_text_properties( - _or_dummy(st.find("style:text-properties", ns)), ns, face_map - ) - if fams: - style_map[name] = fams - return style_map, default_fams - - -def _lookup_style_families( - style_name: str, ns: dict[str, str], face_map: dict[str, str], roots: list[ET.Element | None] -) -> set[str]: - fams: set[str] = set() - if not style_name: - return fams - visited: set[str] = set() - - def _resolve(name: str) -> None: - if not name or name in visited: - return - visited.add(name) - for root in roots: - if root is None: - continue - node = root.find(f".//style:style[@style:name='{name}']", ns) - if node is None: - node = root.find(f".//style:style[@{{{ns['style']}}}name='{name}']", ns) - if node is None: - continue - fams.update( - _families_from_text_properties( - _or_dummy(node.find("style:text-properties", ns)), ns, face_map - ) - ) - # Follow parent style chain if present - parent = node.get( - "{urn:oasis:names:tc:opendocument:xmlns:style:1.0}parent-style-name" - ) or node.get("style:parent-style-name") - if parent: - _resolve(parent) - - _resolve(style_name) - return fams - - -def _collect_slide_families( - page: ET.Element, - ns: dict[str, str], - style_map: dict[str, set[str]], - face_map: dict[str, str], - roots: list[ET.Element | None], - text_style_map: dict[str, set[str]] | None = None, -) -> set[str]: - slide_fams: set[str] = set() - for el in page.iter(): - fam_attr = el.get( - "{urn:oasis:names:tc:opendocument:xmlns:xsl-fo-compatible:1.0}font-family" - ) - if fam_attr: - slide_fams.update(_split_odf_family_list(fam_attr)) - for attr in ( - "{urn:oasis:names:tc:opendocument:xmlns:text:1.0}style-name", - "text:style-name", - "{urn:oasis:names:tc:opendocument:xmlns:drawing:1.0}text-style-name", - "draw:text-style-name", - "draw:style-name", - "presentation:style-name", - ): - style_name = el.get(attr) - if not style_name: - continue - resolved_fams: set[str] = set() - if style_name in style_map: - resolved_fams.update(style_map[style_name]) - if not resolved_fams: - # Fallback: resolve on the fly from XML if not present in prebuilt style_map - resolved_fams.update(_lookup_style_families(style_name, ns, face_map, roots)) - if not resolved_fams and text_style_map and style_name in text_style_map: - resolved_fams.update(text_style_map[style_name]) - if resolved_fams: - slide_fams.update(resolved_fams) - return slide_fams - - -def _build_style_map_text(xml_text: str) -> dict[str, set[str]]: - # Best-effort textual extraction for cases missed by XML namespace lookups - # Finds style:style name="X" blocks and extracts fo:font-family and style:font-name attributes - style_map: dict[str, set[str]] = {} - # Non-greedy match of a style:style block - for m in re.finditer( - r"]*?\bstyle:name=\"([^\"]+)\"[\s\S]*?(?:)", - xml_text, - flags=re.IGNORECASE, - ): - name = m.group(1).strip() - block = m.group(0) - fams: set[str] = set() - # fo:font-family may be a comma list - mff = re.search(r"fo:font-family=\"([^\"]+)\"", block, flags=re.IGNORECASE) - if mff: - for f in _split_odf_family_list(mff.group(1)): - fams.add(f) - # style:font-name may be a face alias; treat as family directly if present - mfn = re.search(r"style:font-name=\"([^\"]+)\"", block, flags=re.IGNORECASE) - if mfn: - fams.add(normalize_font_family_name(mfn.group(1))) - if fams: - style_map[name] = fams - return style_map - - -def _extract_slide_families_from_odp(odp_path: str) -> dict[int, set[str]]: - ns = { - "office": "urn:oasis:names:tc:opendocument:xmlns:office:1.0", - "style": "urn:oasis:names:tc:opendocument:xmlns:style:1.0", - "fo": "urn:oasis:names:tc:opendocument:xmlns:xsl-fo-compatible:1.0", - "draw": "urn:oasis:names:tc:opendocument:xmlns:drawing:1.0", - "text": "urn:oasis:names:tc:opendocument:xmlns:text:1.0", - } - by_slide: dict[int, set[str]] = {} - with ZipFile(odp_path, "r") as zf: - content_bytes = zf.read("content.xml") - styles_bytes = zf.read("styles.xml") if "styles.xml" in zf.namelist() else None - content = ET.fromstring(content_bytes) - styles_root = ET.fromstring(styles_bytes) if styles_bytes is not None else None - styles_text = ( - styles_bytes.decode("utf-8", errors="ignore") if styles_bytes is not None else "" - ) - - face_map: dict[str, str] = {} - face_map.update(_collect_face_map(content, ns)) - if styles_root is not None: - face_map.update(_collect_face_map(styles_root, ns)) - - style_map, default_fams = _build_style_map(content, styles_root, ns, face_map) - # Augment style_map with textual parsing fallback (helps with tricky namespace emissions) - text_style_map: dict[str, set[str]] = {} - if styles_text: - text_style_map = _build_style_map_text(styles_text) - for k, v in text_style_map.items(): - if k not in style_map: - style_map[k] = v - - master_map: dict[str, set[str]] = _build_master_page_map(styles_root, ns, style_map) - - pres = content.find("office:body", ns) - if pres is not None: - pres = pres.find("office:presentation", ns) - if pres is None: - return {} - pages = pres.findall("draw:page", ns) - global_fams: set[str] = set() - for idx, page in enumerate(pages, start=1): - slide_fams = _collect_slide_families( - page, ns, style_map, face_map, [content, styles_root], text_style_map - ) - mp_name = page.get( - "{urn:oasis:names:tc:opendocument:xmlns:drawing:1.0}master-page-name" - ) or page.get("draw:master-page-name") - if mp_name and mp_name in master_map: - slide_fams.update(master_map[mp_name]) - # If theme placeholders like +mn lt are present, augment with defaults - if any(f.startswith("+") for f in slide_fams) and default_fams: - slide_fams.update(default_fams) - if not slide_fams and default_fams: - slide_fams.update(default_fams) - expanded: set[str] = set() - for f in slide_fams: - base, _ = parse_font_family_base_and_styles(f) - expanded.add(f) - expanded.add(base) - expanded.add(base.replace(" ", "")) - by_slide[idx] = expanded - global_fams.update(expanded) - # As a last resort, use global families - if global_fams: - for idx in list(by_slide.keys()): - if not by_slide[idx]: - by_slide[idx] = set(global_fams) - elif all(f.startswith("+") for f in by_slide[idx]): - by_slide[idx].update(global_fams) - return by_slide - - -def _build_master_page_map( - styles_root: ET.Element | None, ns: dict[str, str], style_map: dict[str, set[str]] -) -> dict[str, set[str]]: - master_map: dict[str, set[str]] = {} - if styles_root is None: - return master_map - master_styles = styles_root.find("office:master-styles", ns) - if master_styles is None: - return master_map - for mp in master_styles.findall("draw:master-page", ns): - mname = mp.get("{urn:oasis:names:tc:opendocument:xmlns:drawing:1.0}name") or mp.get( - "draw:name" - ) - if not mname: - continue - fams: set[str] = set() - for el in mp.iter(): - fam_attr = el.get( - "{urn:oasis:names:tc:opendocument:xmlns:xsl-fo-compatible:1.0}font-family" - ) - if fam_attr: - fams.update(_split_odf_family_list(fam_attr)) - for attr in ( - "{urn:oasis:names:tc:opendocument:xmlns:text:1.0}style-name", - "text:style-name", - "{urn:oasis:names:tc:opendocument:xmlns:drawing:1.0}text-style-name", - "draw:text-style-name", - "draw:style-name", - "presentation:style-name", - ): - sname = el.get(attr) - if sname and sname in style_map: - fams.update(style_map[sname]) - if fams: - expanded: set[str] = set() - for f in fams: - base, _ = parse_font_family_base_and_styles(f) - expanded.add(f) - expanded.add(base) - expanded.add(base.replace(" ", "")) - master_map[mname] = expanded - return master_map - - -def detect_missing_fonts_odp(pptx_path: str) -> tuple[set[str], dict[int, list[str]]]: - pptx_path = abspath(pptx_path) - used = extract_used_fonts_from_pptx(pptx_path) - with tempfile.TemporaryDirectory(prefix="soffice_profile_") as prof: - with tempfile.TemporaryDirectory(prefix="soffice_convert_") as out: - stem = splitext(basename(pptx_path))[0] - odp_path = _export_to_odp(pptx_path, prof, out, stem) - if not odp_path: - return set(), {} - slide_fams = _extract_slide_families_from_odp(odp_path) - - missing_overall: set[str] = set() - missing_by_slide: dict[int, list[str]] = {} - syn_map = _build_fc_synonym_map() - for slide_num, req_fams in used.items(): - odp_fams = slide_fams.get(slide_num, set()) - slide_missing: list[str] = [] - for req in req_fams: - fam_base, _ = parse_font_family_base_and_styles(req) - # Accept fontconfig-resolved aliases and no-space variants for the requested base family - acceptable: set[str] = _expand_via_fontconfig(fam_base) - # Determine if any acceptable alias is actually installed on system - installed = any(alias in syn_map for alias in acceptable) - # Missing if not installed at all, or if installed but not resolved in ODP families - if (not installed) or ((req not in odp_fams) and not (acceptable & odp_fams)): - slide_missing.append(req) - missing_overall.add(req) - if slide_missing: - missing_by_slide[slide_num] = sorted(slide_missing) - return missing_overall, missing_by_slide - - -def main() -> None: - parser = argparse.ArgumentParser( - description=( - "Detect missing/substituted fonts for a PPTX by converting to ODP and inspecting resolved families." - ) - ) - parser.add_argument("pptx_path", help="Path to .pptx file") - parser.add_argument( - "--json", dest="output_json", action="store_true", default=False, help="Emit JSON output" - ) - parser.add_argument( - "--include-missing", - dest="include_missing", - action="store_true", - default=True, - help="Include missing category", - ) - parser.add_argument( - "--include-substituted", - dest="include_substituted", - action="store_true", - default=True, - help="Include substituted category", - ) - args = parser.parse_args() - - pptx_path = abspath(expanduser(args.pptx_path)) - used = extract_used_fonts_from_pptx(pptx_path) - # Only build ODP families if we need to report substitutions - slide_fams: dict[int, set[str]] = {} - odp_available = False - if args.include_substituted: - with tempfile.TemporaryDirectory(prefix="soffice_profile_") as prof: - with tempfile.TemporaryDirectory(prefix="soffice_convert_") as out: - stem = splitext(basename(pptx_path))[0] - odp_path = _export_to_odp(pptx_path, prof, out, stem) - if odp_path: - slide_fams = _extract_slide_families_from_odp(odp_path) - odp_available = True - - syn_map = _build_fc_synonym_map() - font_missing_by_slide: dict[int, list[str]] = {} - font_substituted_by_slide: dict[int, list[str]] = {} - for slide_num, req_fams in used.items(): - if args.include_substituted and odp_available: - odp_fams = slide_fams.get(slide_num, set()) - else: - odp_fams = set() - miss_missing: list[str] = [] - miss_sub: list[str] = [] - for req in req_fams: - fam_base, _ = parse_font_family_base_and_styles(req) - acceptable: set[str] = _expand_via_fontconfig(fam_base) - installed = any(alias in syn_map for alias in acceptable) - if args.include_missing and not installed: - miss_missing.append(req) - if ( - args.include_substituted - and odp_available - and installed - and (req not in odp_fams) - and not (acceptable & odp_fams) - ): - miss_sub.append(req) - if miss_missing: - font_missing_by_slide[slide_num] = sorted(miss_missing) - if miss_sub: - font_substituted_by_slide[slide_num] = sorted(miss_sub) - - font_missing_overall: set[str] = ( - set().union(*font_missing_by_slide.values()) if font_missing_by_slide else set() - ) - font_substituted_overall: set[str] = ( - set().union(*font_substituted_by_slide.values()) if font_substituted_by_slide else set() - ) - - if args.output_json: - payload: dict[str, object] = {} - if args.include_missing: - payload["font_missing_overall"] = sorted(font_missing_overall) - payload["font_missing_by_slide"] = {str(k): v for k, v in font_missing_by_slide.items()} - if args.include_substituted: - payload["font_substituted_overall"] = sorted(font_substituted_overall) - payload["font_substituted_by_slide"] = { - str(k): v for k, v in font_substituted_by_slide.items() - } - print(json.dumps(payload)) - else: - any_missing = args.include_missing and bool(font_missing_overall) - any_sub = args.include_substituted and bool(font_substituted_overall) - if any_missing or any_sub: - if any_missing: - print("Fonts missing (not installed):") - print(", ".join(sorted(font_missing_overall))) - for slide_num in sorted(font_missing_by_slide.keys()): - print(f"Slide {slide_num} missing: ", end="") - print(", ".join(font_missing_by_slide[slide_num])) - if any_sub: - print("Fonts substituted (installed but substituted during rendering):") - print(", ".join(sorted(font_substituted_overall))) - for slide_num in sorted(font_substituted_by_slide.keys()): - print(f"Slide {slide_num} substituted: ", end="") - print(", ".join(font_substituted_by_slide[slide_num])) - else: - print("No font issues detected.") - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-slides/scripts/ensure_raster_image.py b/.github/skills/openai-slides/scripts/ensure_raster_image.py deleted file mode 100644 index 0ce3dbcb..00000000 --- a/.github/skills/openai-slides/scripts/ensure_raster_image.py +++ /dev/null @@ -1,202 +0,0 @@ -#!/usr/bin/env python3 -"""Copyright (c) OpenAI. All rights reserved. - -Ensures input images are rasterized, converting to PNG when needed. Primarily used to -preview image assets extracted from PowerPoint files. - - -Dependencies used by this tool: -- Inkscape: SVG/EMF/WMF rasterization -- ImageMagick: format bridging (TIFF→PNG, generic convert) -- Ghostscript: PDF/EPS/PS rasterization (first page) -- libheif-examples: heif-convert for HEIC/HEIF → PNG -- jxr-tools (or libjxr-tools on older distros): JxrDecApp for JPEG XR (JXR/WDP) - -Install (Ubuntu/Debian): - sudo apt-get update - sudo apt-get install -y inkscape imagemagick ghostscript libheif-examples jxr-tools - # If jxr-tools not found on your distro, try: - # sudo apt-get install -y libjxr-tools - -Verify: - inkscape --version - convert -version | grep -i "ImageMagick" - gs -v - heif-convert -h - JxrDecApp -h -""" - -import argparse -import gzip -import shutil -from os import listdir -from os.path import basename, dirname, expanduser, isfile, join, splitext -from subprocess import run - -RASTER_EXTS = { - ".png", - ".jpg", - ".jpeg", - ".bmp", - ".gif", - ".tif", - ".tiff", - ".webp", -} - -CONVERTIBLE_EXTS = { - # Windows metafiles (and compressed variants) - ".emf", - ".wmf", - ".emz", - ".wmz", - # SVG - ".svg", - ".svgz", - # JPEG XR / HD Photo - ".wdp", - ".jxr", - # HEIF family - ".heic", - ".heif", - # Page-description formats (rasterize first page) - ".pdf", - ".eps", - ".ps", -} - -SUPPORTED_EXTS = RASTER_EXTS | CONVERTIBLE_EXTS - - -def _imagemagick_convert(src_path: str, dst_path: str) -> None: - binary = shutil.which("magick") or "convert" - run([binary, src_path, dst_path], check=True) - - -def ensure_raster_image(path: str, out_dir: str | None = None) -> str: - """Return a raster image path for the given input, converting when needed. - - - EMF/WMF/EMZ/WMZ are rasterized via Inkscape (EMZ/WMZ are decompressed first) - - SVG/SVGZ are rasterized via Inkscape - - WDP/JXR are converted via ImageMagick (if codec available) - - Known raster formats are returned as-is - - Raises ValueError if the extension is not supported. - """ - base, ext = splitext(path) - ext_lower = ext.lower() - out_dir = out_dir or dirname(path) - out_path = join(out_dir, basename(base) + ".png") - - # Convertible formats - if ext_lower in (".emf", ".wmf"): - run(["inkscape", path, "-o", out_path], check=True) - if isfile(out_path): - return out_path - raise RuntimeError("inkscape reported success but output file not found: " + out_path) - - if ext_lower in (".emz", ".wmz"): - # Decompress into EMF/WMF then rasterize with Inkscape - decompressed = join(out_dir, basename(base) + (".emf" if ext_lower == ".emz" else ".wmf")) - with gzip.open(path, "rb") as zin, open(decompressed, "wb") as zout: - zout.write(zin.read()) - run( - ["inkscape", decompressed, "-o", out_path], - check=True, - ) - if isfile(out_path): - return out_path - raise RuntimeError("inkscape reported success but output file not found: " + out_path) - - if ext_lower in (".svg", ".svgz"): - run(["inkscape", path, "-o", out_path], check=True) - if isfile(out_path): - return out_path - raise RuntimeError("inkscape reported success but output file not found: " + out_path) - - if ext_lower in (".wdp", ".jxr"): - tmp_tiff = join(out_dir, basename(base) + ".tiff") - run(["JxrDecApp", "-i", path, "-o", tmp_tiff], check=True) - _imagemagick_convert(tmp_tiff, out_path) - if isfile(out_path): - return out_path - raise RuntimeError("JPEG XR decode succeeded but PNG not found: " + out_path) - - if ext_lower in (".heic", ".heif"): - # Use libheif's CLI for robust conversion - heif_convert = shutil.which("heif-convert") or "heif-convert" - run([heif_convert, path, out_path], check=True) - if isfile(out_path): - return out_path - raise RuntimeError("heif-convert reported success but output file not found: " + out_path) - - if ext_lower in (".pdf", ".eps", ".ps"): - # Rasterize first page via Ghostscript - gs = shutil.which("gs") or "gs" - run( - [ - gs, - "-dSAFER", - "-dBATCH", - "-dNOPAUSE", - "-sDEVICE=pngalpha", - "-dFirstPage=1", - "-dLastPage=1", - "-r200", - "-o", - out_path, - path, - ], - check=True, - ) - if isfile(out_path): - return out_path - raise RuntimeError("Ghostscript reported success but output file not found: " + out_path) - - if ext_lower in RASTER_EXTS: - return path - - raise ValueError(f"Unsupported image format for montage: {path}") - - -def main() -> None: - parser = argparse.ArgumentParser( - description=("Ensure input images are rasterized; convert to PNG if needed.") - ) - group = parser.add_mutually_exclusive_group(required=True) - group.add_argument("--input_files", nargs="+", help="List of input image file paths") - group.add_argument("--input_dir", help="Directory containing input images") - parser.add_argument( - "--output_dir", - default=None, - help=( - "Directory to write converted PNGs. If omitted, converted files are written next to inputs." - ), - ) - args = parser.parse_args() - - if args.input_files: - paths = [expanduser(p) for p in args.input_files] - else: - input_dir = expanduser(args.input_dir) - names = listdir(input_dir) - paths = [ - join(input_dir, f) - for f in names - if isfile(join(input_dir, f)) and splitext(f)[1].lower() in SUPPORTED_EXTS - ] - if not paths: - raise SystemExit("No files with supported extensions in input_dir") - - out_dir = expanduser(args.output_dir) if args.output_dir else None - converted_paths = [] - for p in paths: - if ensure_raster_image(p, out_dir) != p: - converted_paths.append(p) - - if converted_paths: - print("Converted the following files to PNG:\n" + "\n".join(converted_paths)) - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-slides/scripts/render_slides.py b/.github/skills/openai-slides/scripts/render_slides.py deleted file mode 100644 index c5402337..00000000 --- a/.github/skills/openai-slides/scripts/render_slides.py +++ /dev/null @@ -1,273 +0,0 @@ -#!/usr/bin/env python3 -# Copyright (c) OpenAI. All rights reserved. -import argparse -import os -import re -import subprocess -import tempfile -import xml.etree.ElementTree as ET -from os import makedirs, replace -from os.path import abspath, basename, exists, expanduser, join, splitext -from typing import Sequence, cast -from zipfile import ZipFile - -from pdf2image import convert_from_path, pdfinfo_from_path - -EMU_PER_INCH: int = 914_400 - - -def calc_dpi_via_ooxml(input_path: str, max_w_px: int, max_h_px: int) -> int: - """Calculate DPI from OOXML `ppt/presentation.xml` slide size (cx/cy in EMUs).""" - with ZipFile(input_path, "r") as zf: - xml = zf.read("ppt/presentation.xml") - root = ET.fromstring(xml) - ns = {"p": "http://schemas.openxmlformats.org/presentationml/2006/main"} - sld_sz = root.find("p:sldSz", ns) - if sld_sz is None: - raise RuntimeError("Slide size not found in presentation.xml") - cx = int(sld_sz.get("cx") or 0) - cy = int(sld_sz.get("cy") or 0) - if cx <= 0 or cy <= 0: - raise RuntimeError("Invalid slide size values in presentation.xml") - width_in = cx / EMU_PER_INCH - height_in = cy / EMU_PER_INCH - return round(min(max_w_px / width_in, max_h_px / height_in)) - - -def calc_dpi_via_pdf(input_path: str, max_w_px: int, max_h_px: int) -> int: - """Compute DPI from PDF page size. - - For non-PDF inputs, first convert to PDF via LibreOffice to read page size. - For PDFs, use the PDF directly (avoids unnecessary conversion and failures). - """ - is_pdf = input_path.lower().endswith(".pdf") - with tempfile.TemporaryDirectory(prefix="soffice_profile_") as user_profile: - with tempfile.TemporaryDirectory(prefix="soffice_convert_") as convert_tmp_dir: - stem = splitext(basename(input_path))[0] - pdf_path = ( - input_path - if is_pdf - else convert_to_pdf(input_path, user_profile, convert_tmp_dir, stem) - ) - if not (pdf_path and exists(pdf_path)): - raise RuntimeError("Failed to produce/read PDF for DPI computation.") - - info = pdfinfo_from_path(pdf_path) - size_val = info.get("Page size") - if not size_val: - for k, v in info.items(): - if isinstance(v, str) and "size" in k.lower() and "pts" in v: - size_val = v - break - if not isinstance(size_val, str): - raise RuntimeError("Failed to read PDF page size for DPI computation.") - - def _parse_page_size_to_pts(s: str) -> tuple[float, float]: - # Common formats from poppler/pdfinfo: - # - "612 x 792 pts (letter)" - # - "595.276 x 841.89 pts (A4)" - # - sometimes inches: "8.5 x 11 in" - m_pts = re.search( - r"([0-9]+(?:\.[0-9]+)?)\s*x\s*([0-9]+(?:\.[0-9]+)?)\s*pts\b", - s, - ) - if m_pts: - return float(m_pts.group(1)), float(m_pts.group(2)) - m_in = re.search( - r"([0-9]+(?:\.[0-9]+)?)\s*x\s*([0-9]+(?:\.[0-9]+)?)\s*in\b", - s, - ) - if m_in: - w_in = float(m_in.group(1)) - h_in = float(m_in.group(2)) - return w_in * 72.0, h_in * 72.0 - # Sometimes poppler returns without an explicit unit; treat as points. - m = re.search(r"([0-9]+(?:\.[0-9]+)?)\s*x\s*([0-9]+(?:\.[0-9]+)?)\b", s) - if m: - return float(m.group(1)), float(m.group(2)) - raise RuntimeError(f"Unrecognized PDF page size format: {s!r}") - - width_pts, height_pts = _parse_page_size_to_pts(size_val) - width_in = width_pts / 72.0 - height_in = height_pts / 72.0 - if width_in <= 0 or height_in <= 0: - raise RuntimeError("Invalid PDF page size values.") - return round(min(max_w_px / width_in, max_h_px / height_in)) - - -def run_cmd_no_check(cmd: list[str]) -> None: - subprocess.run( - cmd, - check=False, - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - env=os.environ.copy(), - ) - - -def convert_to_pdf( - pptx_path: str, - user_profile: str, - convert_tmp_dir: str, - stem: str, -) -> str: - # Try direct PPTX -> PDF - cmd_pdf = [ - "soffice", - "-env:UserInstallation=file://" + user_profile, - "--invisible", - "--headless", - "--norestore", - "--convert-to", - "pdf", - "--outdir", - convert_tmp_dir, - pptx_path, - ] - run_cmd_no_check(cmd_pdf) - - pdf_path = join(convert_tmp_dir, f"{stem}.pdf") - if exists(pdf_path): - return pdf_path - - # Fallback: PPTX -> ODP, then ODP -> PDF - # Rationale: Saving as ODP normalizes PPTX-specific constructs via the ODF serializer, - # which often bypasses Impress PDF export issues on problematic decks. - cmd_odp = [ - "soffice", - "-env:UserInstallation=file://" + user_profile, - "--invisible", - "--headless", - "--norestore", - "--convert-to", - "odp", - "--outdir", - convert_tmp_dir, - pptx_path, - ] - run_cmd_no_check(cmd_odp) - - odp_path = join(convert_tmp_dir, f"{stem}.odp") - - if exists(odp_path): - # ODP -> PDF - cmd_odp_pdf = [ - "soffice", - "-env:UserInstallation=file://" + user_profile, - "--invisible", - "--headless", - "--norestore", - "--convert-to", - "pdf", - "--outdir", - convert_tmp_dir, - odp_path, - ] - run_cmd_no_check(cmd_odp_pdf) - if exists(pdf_path): - return pdf_path - - return "" - - -def rasterize( - input_path: str, - out_dir: str, - dpi: int, -) -> Sequence[str]: - """Rasterise PPTX/PDF to PNG files placed in out_dir and return the image paths.""" - makedirs(out_dir, exist_ok=True) - input_path = abspath(input_path) - stem = splitext(basename(input_path))[0] - - # Use a unique user profile to avoid LibreOffice profile lock when running concurrently - with tempfile.TemporaryDirectory(prefix="soffice_profile_") as user_profile: - # Write conversion outputs into a temp directory to avoid any IO oddities - with tempfile.TemporaryDirectory(prefix="soffice_convert_") as convert_tmp_dir: - is_pdf = input_path.lower().endswith(".pdf") - pdf_path = ( - input_path - if is_pdf - else convert_to_pdf(input_path, user_profile, convert_tmp_dir, stem) - ) - - if not pdf_path or not exists(pdf_path): - raise RuntimeError( - "Failed to produce PDF for rasterization (direct and ODP fallback)." - ) - - # Perform rasterization while the temp PDF still exists - paths_raw = cast( - list[str], - convert_from_path( - pdf_path, - dpi=dpi, - fmt="png", - thread_count=8, - output_folder=out_dir, - paths_only=True, - output_file="slide", - ), - ) - # Rename convert_from_path's output format f'slide{thread_id:04d}-{page_num:02d}.png' - slides = [] - for src_path in paths_raw: - base = splitext(basename(src_path))[0] - slide_num_str = base.split("-")[-1] - slide_num = int(slide_num_str) - dst_path = join(out_dir, f"slide-{slide_num}.png") - replace(src_path, dst_path) - slides.append((slide_num, dst_path)) - slides.sort(key=lambda t: t[0]) - final_paths = [path for _, path in slides] - return final_paths - - -def main() -> None: - parser = argparse.ArgumentParser(description="Render slides to images.") - parser.add_argument( - "input_path", - type=str, - help="Path to the input PowerPoint or PDF file.", - ) - parser.add_argument( - "--output_dir", - type=str, - default=None, - help=( - "Output directory for the rendered images. " - "Defaults to a folder next to the input named after the input file (without extension)." - ), - ) - parser.add_argument( - "--width", - type=int, - default=1600, - help=( - "Approximate maximum width in pixels after isotropic scaling (default 1600). " - "The actual value may exceed slightly." - ), - ) - parser.add_argument( - "--height", - type=int, - default=900, - help=( - "Approximate maximum height in pixels after isotropic scaling (default 900). " - "The actual value may exceed slightly." - ), - ) - args = parser.parse_args() - - input_path = abspath(expanduser(args.input_path)) - out_dir = abspath(expanduser(args.output_dir)) if args.output_dir else splitext(input_path)[0] - if input_path.lower().endswith((".pptx", ".ppsx", ".potx", ".pptm", ".ppsm", ".potm")): - dpi = calc_dpi_via_ooxml(input_path, args.width, args.height) - else: - dpi = calc_dpi_via_pdf(input_path, args.width, args.height) - rasterize(input_path, out_dir, dpi) - print("Slides rendered to " + out_dir) - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-slides/scripts/slides_test.py b/.github/skills/openai-slides/scripts/slides_test.py deleted file mode 100644 index 1721681c..00000000 --- a/.github/skills/openai-slides/scripts/slides_test.py +++ /dev/null @@ -1,201 +0,0 @@ -#!/usr/bin/env python3 -# Copyright (c) OpenAI. All rights reserved. -import argparse -import sys -import tempfile -from os.path import abspath, expanduser, join -from pathlib import Path -from typing import Sequence, cast - -import numpy as np - -SCRIPT_DIR = Path(__file__).resolve().parent -if str(SCRIPT_DIR) not in sys.path: - sys.path.insert(0, str(SCRIPT_DIR)) - -import render_slides # type: ignore -from PIL import Image -from pptx import Presentation -from pptx.dml.color import RGBColor -from pptx.enum.shapes import MSO_AUTO_SHAPE_TYPE -from pptx.util import Emu - -# Configuration specific to overflow checking -PAD_PX: int = 100 # fixed padding on every side in pixels -PAD_RGB = (200, 200, 200) -EMU_PER_INCH: int = 914_400 - - -def px_to_emu(px: int, dpi: int) -> Emu: - return Emu(int(px * EMU_PER_INCH // dpi)) - - -def calc_tol(dpi: int) -> int: - """Calculate per-channel colour tolerance appropriate for *dpi* (anti-aliasing tolerance).""" - if dpi >= 300: - return 0 - # 1 at 250 DPI, 5 at 150 DPI, capped to 10. - tol = round((300 - dpi) / 25) - return min(max(tol, 1), 10) - - -def enlarge_deck(src: str, dst: str, pad_emu: Emu) -> tuple[int, int]: - """Enlarge the input PPTX with a fixed grey padding and return the new page size.""" - prs = Presentation(src) - w0 = cast(Emu, prs.slide_width) - h0 = cast(Emu, prs.slide_height) - w1 = Emu(w0 + 2 * pad_emu) - h1 = Emu(h0 + 2 * pad_emu) - prs.slide_width = w1 - prs.slide_height = h1 - - for slide in prs.slides: - # Shift all shapes so the original canvas sits centred in the new deck. - for shp in list(slide.shapes): - shp.left = Emu(int(shp.left) + pad_emu) - shp.top = Emu(int(shp.top) + pad_emu) - - pads = ( - (Emu(0), Emu(0), pad_emu, h1), # left - (Emu(int(w1) - int(pad_emu)), Emu(0), pad_emu, h1), # right - (Emu(0), Emu(0), w1, pad_emu), # top - (Emu(0), Emu(int(h1) - int(pad_emu)), w1, pad_emu), # bottom - ) - - sp_tree = slide.shapes._spTree # pylint: disable=protected-access - - for left, top, width, height in pads: - pad_shape = slide.shapes.add_shape( - MSO_AUTO_SHAPE_TYPE.RECTANGLE, left, top, width, height - ) - pad_shape.fill.solid() - pad_shape.fill.fore_color.rgb = RGBColor(*PAD_RGB) - pad_shape.line.fill.background() - - # Send pad behind all other shapes (index 2 after mandatory nodes) - sp_tree.remove(pad_shape._element) - sp_tree.insert(2, pad_shape._element) - - prs.save(dst) - return int(w1), int(h1) - - -def inspect_images( - paths: Sequence[str], - pad_ratio_w: float, - pad_ratio_h: float, - dpi: int, -) -> list[int]: - """Return 1-based indices of slides that contain pixels outside the pad.""" - - tol = calc_tol(dpi) - failures: list[int] = [] - pad_colour = np.array(PAD_RGB, dtype=np.uint8) - - for idx, img_path in enumerate(paths, start=1): - with Image.open(img_path) as img: - rgb = img.convert("RGB") - arr = np.asarray(rgb) - - h, w, _ = arr.shape - # Exclude the innermost 1-pixel band - pad_x = int(w * pad_ratio_w) - 1 - pad_y = int(h * pad_ratio_h) - 1 - - left_margin = arr[:, :pad_x, :] - right_margin = arr[:, w - pad_x :, :] - top_margin = arr[:pad_y, :, :] - bottom_margin = arr[h - pad_y :, :, :] - - def _is_clean(margin: np.ndarray) -> bool: - diff = np.abs(margin.astype(np.int16) - pad_colour) - matches = np.all(diff <= tol, axis=-1) - mismatch_fraction = 1.0 - (np.count_nonzero(matches) / matches.size) - if dpi >= 300: - max_mismatch = 0.01 - elif dpi >= 200: - max_mismatch = 0.02 - else: - max_mismatch = 0.03 - return mismatch_fraction <= max_mismatch - - if not ( - _is_clean(left_margin) - and _is_clean(right_margin) - and _is_clean(top_margin) - and _is_clean(bottom_margin) - ): - failures.append(idx) - - return failures - - -def main() -> None: - parser = argparse.ArgumentParser( - description=( - "Check a PPTX for content overflowing the original canvas by rendering with padding " - "and inspecting the margins." - ) - ) - parser.add_argument( - "input_path", - type=str, - help="Path to the input PPTX file.", - ) - parser.add_argument( - "--width", - type=int, - default=1600, - help=( - "Approximate maximum width in pixels after isotropic scaling (default 1600). " - "The actual value may exceed slightly." - ), - ) - parser.add_argument( - "--height", - type=int, - default=900, - help=( - "Approximate maximum height in pixels after isotropic scaling (default 900). " - "The actual value may exceed slightly." - ), - ) - parser.add_argument( - "--pad_px", - type=int, - default=PAD_PX, - help="Padding in pixels to add on each side before rasterization.", - ) - args = parser.parse_args() - - input_path = abspath(expanduser(args.input_path)) - # Width and height refer to the original, unaltered slide dimensions. - dpi = render_slides.calc_dpi_via_ooxml(input_path, args.width, args.height) - - # Not using ``tempfile.TemporaryDirectory(delete=False)`` for Python 3.11 compatibility. - tmpdir = tempfile.mkdtemp() - enlarged_pptx = join(tmpdir, "enlarged.pptx") - pad_emu = px_to_emu(args.pad_px, dpi) - w1, h1 = enlarge_deck(input_path, enlarged_pptx, pad_emu=pad_emu) - pad_ratio_w = pad_emu / w1 - pad_ratio_h = pad_emu / h1 - - img_dir = join(tmpdir, "imgs") - img_paths = render_slides.rasterize(enlarged_pptx, img_dir, dpi) - failing = inspect_images(img_paths, pad_ratio_w, pad_ratio_h, dpi) - - if failing: - print( - "ERROR: Slides with content overflowing original canvas (1-based indexing): " - + ", ".join(map(str, failing)) - + "\n" - + "Rendered images with grey paddings for problematic slides are available at: " - ) - for i in failing: - print(img_paths[i - 1]) - else: - print("Test passed. No overflow detected.") - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-spreadsheet/LICENSE.txt b/.github/skills/openai-spreadsheet/LICENSE.txt deleted file mode 100644 index 13e25df8..00000000 --- a/.github/skills/openai-spreadsheet/LICENSE.txt +++ /dev/null @@ -1,201 +0,0 @@ -Apache License -Version 2.0, January 2004 -http://www.apache.org/licenses/ - -TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - -1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - -2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - -3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - -4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - -5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - -6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - -7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - -8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - -9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf of - any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - -END OF TERMS AND CONDITIONS - -APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don\'t include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - -Copyright [yyyy] [name of copyright owner] - -Licensed under the Apache License, Version 2.0 (the "License"); -you may not use this file except in compliance with the License. -You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - -Unless required by applicable law or agreed to in writing, software -distributed under the License is distributed on an "AS IS" BASIS, -WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -See the License for the specific language governing permissions and -limitations under the License. diff --git a/.github/skills/openai-spreadsheet/SKILL.md b/.github/skills/openai-spreadsheet/SKILL.md deleted file mode 100644 index 22369087..00000000 --- a/.github/skills/openai-spreadsheet/SKILL.md +++ /dev/null @@ -1,158 +0,0 @@ ---- -name: "openai-spreadsheet" -description: "Use when tasks involve creating, editing, analyzing, or formatting spreadsheets (`.xlsx`, `.csv`, `.tsv`) with formula-aware workflows, cached recalculation, and visual review." ---- - -# Spreadsheet Skill - -## When to use -- Create new workbooks with formulas, formatting, and structured layouts. -- Read or analyze tabular data (filter, aggregate, pivot, compute metrics). -- Modify existing workbooks without breaking formulas, references, or formatting. -- Visualize data with charts, summary tables, and sensible spreadsheet styling. -- Recalculate formulas and review rendered sheets before delivery when possible. - -IMPORTANT: System and user instructions always take precedence. - -## Workflow -1. Confirm the file type and goal: create, edit, analyze, or visualize. -2. Prefer `openpyxl` for `.xlsx` editing and formatting. Use `pandas` for analysis and CSV/TSV workflows. -3. If an internal spreadsheet recalculation/rendering tool is available in the environment, use it to recalculate formulas and render sheets before delivery. -4. Use formulas for derived values instead of hardcoding results. -5. If layout matters, render for visual review and inspect the output. -6. Save outputs, keep filenames stable, and clean up intermediate files. - -## Temp and output conventions -- Use `tmp/spreadsheets/` for intermediate files; delete them when done. -- Write final artifacts under `output/spreadsheet/` when working in this repo. -- Keep filenames stable and descriptive. - -## Primary tooling -- Use `openpyxl` for creating/editing `.xlsx` files and preserving formatting. -- Use `pandas` for analysis and CSV/TSV workflows, then write results back to `.xlsx` or `.csv`. -- Use `openpyxl.chart` for native Excel charts when needed. -- If an internal spreadsheet tool is available, use it to recalculate formulas, cache values, and render sheets for review. - -## Structured Data Evidence Budget -- For large `.xlsx`, `.csv`, and `.tsv` work, keep user-facing evidence compact: - report schema or headers, row counts, column counts, targeted anomalies, - checksums or hashes when useful, sampled examples, and validation gaps. -- Start discovery with headers plus a small sample, then move to deterministic - full-file checks when correctness depends on the whole dataset. -- Sampling does not replace full-file validation for transforms, merges, - source-link checks, empty-row checks, column moves, stable ID generation, - duplicate ID detection, or reconciliation. -- Preserve source links, formulas, formatting where applicable, empty-row - anomalies, duplicate IDs, stable generated IDs, and missing column data as - material integrity checks. - -## Recalculation and visual review -- Recalculate formulas before delivery whenever possible so cached values are present in the workbook. -- Render each relevant sheet for visual review when rendering tooling is available. -- `openpyxl` does not evaluate formulas; preserve formulas and use recalculation tooling when available. -- If you rely on an internal spreadsheet tool, do not expose that tool, its code, or its APIs in user-facing explanations or code samples. - -## Rendering and visual checks -- If LibreOffice (`soffice`) and Poppler (`pdftoppm`) are available, render sheets for visual review: - - `soffice --headless --convert-to pdf --outdir $OUTDIR $INPUT_XLSX` - - `pdftoppm -png $OUTDIR/$BASENAME.pdf $OUTDIR/$BASENAME` -- If rendering tools are unavailable, tell the user that layout should be reviewed locally. -- Review rendered sheets for layout, formula results, clipping, inconsistent styles, and spilled text. - -## Dependencies (install if missing) -Prefer `uv` for dependency management. - -Python packages: -``` -uv pip install openpyxl pandas -``` -If `uv` is unavailable: -``` -python3 -m pip install openpyxl pandas -``` -Optional: -``` -uv pip install matplotlib -``` -If `uv` is unavailable: -``` -python3 -m pip install matplotlib -``` -System tools (for rendering): -``` -# macOS (Homebrew) -brew install libreoffice poppler - -# Ubuntu/Debian -sudo apt-get install -y libreoffice poppler-utils -``` - -If installation is not possible in this environment, tell the user which dependency is missing and how to install it locally. - -## Environment -No required environment variables. - -## Examples -- Runnable Codex examples (openpyxl): `references/examples/openpyxl/` - -## Formula requirements -- Use formulas for derived values rather than hardcoding results. -- Do not use dynamic array functions like `FILTER`, `XLOOKUP`, `SORT`, or `SEQUENCE`. -- Keep formulas simple and legible; use helper cells for complex logic. -- Avoid volatile functions like `INDIRECT` and `OFFSET` unless required. -- Prefer cell references over magic numbers (for example, `=H6*(1+$B$3)` instead of `=H6*1.04`). -- Use absolute (`$B$4`) or relative (`B4`) references carefully so copied formulas behave correctly. -- If you need literal text that starts with `=`, prefix it with a single quote. -- Guard against `#REF!`, `#DIV/0!`, `#VALUE!`, `#N/A`, and `#NAME?` errors. -- Check for off-by-one mistakes, circular references, and incorrect ranges. - -## Citation requirements -- Cite sources inside the spreadsheet using plain-text URLs. -- For financial models, cite model inputs in cell comments. -- For tabular data sourced externally, add a source column when each row represents a separate item. - -## Formatting requirements (existing formatted spreadsheets) -- Render and inspect a provided spreadsheet before modifying it when possible. -- Preserve existing formatting and style exactly. -- Match styles for any newly filled cells that were previously blank. -- Never overwrite established formatting unless the user explicitly asks for a redesign. - -## Formatting requirements (new or unstyled spreadsheets) -- Use appropriate number and date formats. -- Dates should render as dates, not plain numbers. -- Percentages should usually default to one decimal place unless the data calls for something else. -- Currencies should use the appropriate currency format. -- Headers should be visually distinct from raw inputs and derived cells. -- Use fill colors, borders, spacing, and merged cells sparingly and intentionally. -- Set row heights and column widths so content is readable without excessive whitespace. -- Do not apply borders around every filled cell. -- Group related calculations and make totals simple sums of the cells above them. -- Add whitespace to separate sections. -- Ensure text does not spill into adjacent cells. -- Avoid unsupported spreadsheet data-table features such as `=TABLE`. - -## Color conventions (if no style guidance) -- Blue: user input -- Black: formulas and derived values -- Green: linked or imported values -- Gray: static constants -- Orange: review or caution -- Light red: error or flag -- Purple: control or logic -- Teal: visualization anchors and KPI highlights - -## Finance-specific requirements -- Format zeros as `-`. -- Negative numbers should be red and in parentheses. -- Format multiples as `5.2x`. -- Always specify units in headers (for example, `Revenue ($mm)`). -- Cite sources for all raw inputs in cell comments. -- For new financial models with no user-specified style, use blue text for hardcoded inputs, black for formulas, green for internal workbook links, red for external links, and yellow fill for key assumptions that need attention. - -## Investment banking layouts -If the spreadsheet is an IB-style model (LBO, DCF, 3-statement, valuation): -- Totals should sum the range directly above. -- Hide gridlines and use horizontal borders above totals across relevant columns. -- Section headers should be merged cells with dark fill and white text. -- Column labels for numeric data should be right-aligned; row labels should be left-aligned. -- Indent submetrics under their parent line items. diff --git a/.github/skills/openai-spreadsheet/agents/openai.yaml b/.github/skills/openai-spreadsheet/agents/openai.yaml deleted file mode 100644 index c4a670e3..00000000 --- a/.github/skills/openai-spreadsheet/agents/openai.yaml +++ /dev/null @@ -1,6 +0,0 @@ -interface: - display_name: "Spreadsheet Skill" - short_description: "Create, edit, and analyze spreadsheets" - icon_small: "./assets/spreadsheet-small.svg" - icon_large: "./assets/spreadsheet.png" - default_prompt: "Use $spreadsheet to create or update a spreadsheet for this task with the right formulas, structure, and formatting." diff --git a/.github/skills/openai-spreadsheet/assets/spreadsheet-small.svg b/.github/skills/openai-spreadsheet/assets/spreadsheet-small.svg deleted file mode 100644 index c045554b..00000000 --- a/.github/skills/openai-spreadsheet/assets/spreadsheet-small.svg +++ /dev/null @@ -1,3 +0,0 @@ - - - diff --git a/.github/skills/openai-spreadsheet/assets/spreadsheet.png b/.github/skills/openai-spreadsheet/assets/spreadsheet.png deleted file mode 100644 index bd24ca40..00000000 Binary files a/.github/skills/openai-spreadsheet/assets/spreadsheet.png and /dev/null differ diff --git a/.github/skills/openai-spreadsheet/references/examples/openpyxl/create_basic_spreadsheet.py b/.github/skills/openai-spreadsheet/references/examples/openpyxl/create_basic_spreadsheet.py deleted file mode 100644 index b4c58898..00000000 --- a/.github/skills/openai-spreadsheet/references/examples/openpyxl/create_basic_spreadsheet.py +++ /dev/null @@ -1,51 +0,0 @@ -"""Create a basic spreadsheet with two sheets and a simple formula. - -Usage: - python3 create_basic_spreadsheet.py --output /tmp/basic_spreadsheet.xlsx -""" - -from __future__ import annotations - -import argparse -from pathlib import Path - -from openpyxl import Workbook -from openpyxl.utils import get_column_letter - - -def main() -> None: - parser = argparse.ArgumentParser(description="Create a basic spreadsheet with example data.") - parser.add_argument( - "--output", - type=Path, - default=Path("basic_spreadsheet.xlsx"), - help="Output .xlsx path (default: basic_spreadsheet.xlsx)", - ) - args = parser.parse_args() - - wb = Workbook() - overview = wb.active - overview.title = "Overview" - employees = wb.create_sheet("Employees") - - overview["A1"] = "Description" - overview["A2"] = "Awesome Company Report" - - employees.append(["Title", "Name", "Address", "Score"]) - employees.append(["Engineer", "Vicky", "90 50th Street", 98]) - employees.append(["Manager", "Alex", "500 Market Street", 92]) - employees.append(["Designer", "Jordan", "200 Pine Street", 88]) - - employees["A6"] = "Total Score" - employees["D6"] = "=SUM(D2:D4)" - - for col in range(1, 5): - employees.column_dimensions[get_column_letter(col)].width = 20 - - args.output.parent.mkdir(parents=True, exist_ok=True) - wb.save(args.output) - print(f"Saved workbook to {args.output}") - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-spreadsheet/references/examples/openpyxl/create_spreadsheet_with_styling.py b/.github/skills/openai-spreadsheet/references/examples/openpyxl/create_spreadsheet_with_styling.py deleted file mode 100644 index 7ff951cd..00000000 --- a/.github/skills/openai-spreadsheet/references/examples/openpyxl/create_spreadsheet_with_styling.py +++ /dev/null @@ -1,96 +0,0 @@ -"""Generate a styled games scoreboard workbook using openpyxl. - -Usage: - python3 create_spreadsheet_with_styling.py --output /tmp/GamesSimpleStyling.xlsx -""" - -from __future__ import annotations - -import argparse -from pathlib import Path - -from openpyxl import Workbook -from openpyxl.formatting.rule import FormulaRule -from openpyxl.styles import Alignment, Font, PatternFill -from openpyxl.utils import get_column_letter - -HEADER_FILL_HEX = "B7E1CD" -HIGHLIGHT_FILL_HEX = "FFF2CC" - - -def apply_header_style(cell, fill_hex: str) -> None: - cell.fill = PatternFill("solid", fgColor=fill_hex) - cell.font = Font(bold=True) - cell.alignment = Alignment(horizontal="center", vertical="center") - - -def apply_highlight_style(cell, fill_hex: str) -> None: - cell.fill = PatternFill("solid", fgColor=fill_hex) - cell.font = Font(bold=True) - cell.alignment = Alignment(horizontal="center", vertical="center") - - -def populate_game_sheet(ws) -> None: - ws.title = "GameX" - ws.row_dimensions[2].height = 24 - - widths = {"B": 18, "C": 14, "D": 14, "E": 14, "F": 40} - for col, width in widths.items(): - ws.column_dimensions[col].width = width - - headers = ["", "Name", "Game 1 Score", "Game 2 Score", "Total Score", "Notes", ""] - for idx, value in enumerate(headers, start=1): - cell = ws.cell(row=2, column=idx, value=value) - if value: - apply_header_style(cell, HEADER_FILL_HEX) - - players = [ - ("Vicky", 12, 30, "Dominated the minigames."), - ("Yash", 20, 10, "Emily main with strong defense."), - ("Bobby", 1000, 1030, "Numbers look suspiciously high."), - ] - for row_idx, (name, g1, g2, note) in enumerate(players, start=3): - ws.cell(row=row_idx, column=2, value=name) - ws.cell(row=row_idx, column=3, value=g1) - ws.cell(row=row_idx, column=4, value=g2) - ws.cell(row=row_idx, column=5, value=f"=SUM(C{row_idx}:D{row_idx})") - ws.cell(row=row_idx, column=6, value=note) - - ws.cell(row=7, column=2, value="Winner") - ws.cell(row=7, column=3, value="=INDEX(B3:B5, MATCH(MAX(E3:E5), E3:E5, 0))") - ws.cell(row=7, column=5, value="Congrats!") - - ws.merge_cells("C7:D7") - for col in range(2, 6): - apply_highlight_style(ws.cell(row=7, column=col), HIGHLIGHT_FILL_HEX) - - rule = FormulaRule(formula=["LEN(A2)>0"], fill=PatternFill("solid", fgColor=HEADER_FILL_HEX)) - ws.conditional_formatting.add("A2:G2", rule) - - -def main() -> None: - parser = argparse.ArgumentParser(description="Create a styled games scoreboard workbook.") - parser.add_argument( - "--output", - type=Path, - default=Path("GamesSimpleStyling.xlsx"), - help="Output .xlsx path (default: GamesSimpleStyling.xlsx)", - ) - args = parser.parse_args() - - wb = Workbook() - ws = wb.active - populate_game_sheet(ws) - - for col in range(1, 8): - col_letter = get_column_letter(col) - if col_letter not in ws.column_dimensions: - ws.column_dimensions[col_letter].width = 12 - - args.output.parent.mkdir(parents=True, exist_ok=True) - wb.save(args.output) - print(f"Saved workbook to {args.output}") - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-spreadsheet/references/examples/openpyxl/read_existing_spreadsheet.py b/.github/skills/openai-spreadsheet/references/examples/openpyxl/read_existing_spreadsheet.py deleted file mode 100644 index fd0df408..00000000 --- a/.github/skills/openai-spreadsheet/references/examples/openpyxl/read_existing_spreadsheet.py +++ /dev/null @@ -1,59 +0,0 @@ -"""Read an existing .xlsx and print a small summary. - -If --input is not provided, this script creates a tiny sample workbook in /tmp -and reads that instead. -""" - -from __future__ import annotations - -import argparse -import tempfile -from pathlib import Path - -from openpyxl import Workbook, load_workbook - - -def create_sample(path: Path) -> Path: - wb = Workbook() - ws = wb.active - ws.title = "Sample" - ws.append(["Item", "Qty", "Price"]) - ws.append(["Apples", 3, 1.25]) - ws.append(["Oranges", 2, 0.95]) - ws.append(["Bananas", 5, 0.75]) - ws["D1"] = "Total" - ws["D2"] = "=B2*C2" - ws["D3"] = "=B3*C3" - ws["D4"] = "=B4*C4" - wb.save(path) - return path - - -def main() -> None: - parser = argparse.ArgumentParser(description="Read an existing spreadsheet.") - parser.add_argument("--input", type=Path, help="Path to an .xlsx file") - args = parser.parse_args() - - if args.input: - input_path = args.input - else: - tmp_dir = Path(tempfile.gettempdir()) - input_path = tmp_dir / "sample_read_existing.xlsx" - create_sample(input_path) - - wb = load_workbook(input_path, data_only=False) - print(f"Loaded: {input_path}") - print("Sheet names:", wb.sheetnames) - - for name in wb.sheetnames: - ws = wb[name] - max_row = ws.max_row or 0 - max_col = ws.max_column or 0 - print(f"\n== {name} (rows: {max_row}, cols: {max_col})") - for row in ws.iter_rows(min_row=1, max_row=min(max_row, 5), max_col=min(max_col, 5)): - values = [cell.value for cell in row] - print(values) - - -if __name__ == "__main__": - main() diff --git a/.github/skills/openai-spreadsheet/references/examples/openpyxl/styling_spreadsheet.py b/.github/skills/openai-spreadsheet/references/examples/openpyxl/styling_spreadsheet.py deleted file mode 100644 index 37dacdfb..00000000 --- a/.github/skills/openai-spreadsheet/references/examples/openpyxl/styling_spreadsheet.py +++ /dev/null @@ -1,79 +0,0 @@ -"""Create a styled spreadsheet with headers, borders, and a total row. - -Usage: - python3 styling_spreadsheet.py --output /tmp/styling_spreadsheet.xlsx -""" - -from __future__ import annotations - -import argparse -from pathlib import Path - -from openpyxl import Workbook -from openpyxl.styles import Alignment, Border, Font, PatternFill, Side - - -def main() -> None: - parser = argparse.ArgumentParser(description="Create a styled spreadsheet example.") - parser.add_argument( - "--output", - type=Path, - default=Path("styling_spreadsheet.xlsx"), - help="Output .xlsx path (default: styling_spreadsheet.xlsx)", - ) - args = parser.parse_args() - - wb = Workbook() - ws = wb.active - ws.title = "FirstGame" - - ws.merge_cells("B2:E2") - ws["B2"] = "Name | Game 1 Score | Game 2 Score | Total Score" - - header_fill = PatternFill("solid", fgColor="B7E1CD") - header_font = Font(bold=True) - header_alignment = Alignment(horizontal="center", vertical="center") - ws["B2"].fill = header_fill - ws["B2"].font = header_font - ws["B2"].alignment = header_alignment - - ws["B3"] = "Vicky" - ws["C3"] = 50 - ws["D3"] = 60 - ws["E3"] = "=C3+D3" - - ws["B4"] = "John" - ws["C4"] = 40 - ws["D4"] = 50 - ws["E4"] = "=C4+D4" - - ws["B5"] = "Jane" - ws["C5"] = 30 - ws["D5"] = 40 - ws["E5"] = "=C5+D5" - - ws["B6"] = "Jim" - ws["C6"] = 20 - ws["D6"] = 30 - ws["E6"] = "=C6+D6" - - ws.merge_cells("B9:E9") - ws["B9"] = "=SUM(E3:E6)" - - thin = Side(style="thin") - border = Border(top=thin, bottom=thin, left=thin, right=thin) - ws["B9"].border = border - ws["B9"].alignment = Alignment(horizontal="center") - ws["B9"].font = Font(bold=True) - - for col in ("B", "C", "D", "E"): - ws.column_dimensions[col].width = 18 - ws.row_dimensions[2].height = 24 - - args.output.parent.mkdir(parents=True, exist_ok=True) - wb.save(args.output) - print(f"Saved workbook to {args.output}") - - -if __name__ == "__main__": - main() diff --git a/.github/skills/search-company-knowledge/SKILL.md b/.github/skills/search-company-knowledge/SKILL.md new file mode 100644 index 00000000..61f42ed0 --- /dev/null +++ b/.github/skills/search-company-knowledge/SKILL.md @@ -0,0 +1,575 @@ +--- +name: search-company-knowledge +description: "Search across company knowledge bases (Confluence, Jira, internal docs) to find and explain internal concepts, processes, and technical details. When an agent needs to: (1) Find or search for information about systems, terminology, processes, deployment, authentication, infrastructure, architecture, or technical concepts, (2) Search internal documentation, knowledge base, company docs, or our docs, (3) Explain what something is, how it works, or look up information, or (4) Synthesize information from multiple sources. Searches in parallel and provides cited answers." +--- + +# Search Company Knowledge + +## Keywords +find information, search company knowledge, look up, what is, explain, company docs, internal documentation, Confluence search, Jira search, our documentation, internal knowledge, knowledge base, search for, tell me about, get information about, company systems, terminology, find everything about, what do we know about, deployment, authentication, infrastructure, processes, procedures, how to, how does, our systems, our processes, internal systems, company processes, technical documentation, engineering docs, architecture, configuration, search our docs, search internal docs, find in our docs + +## Overview + +Search across siloed company knowledge systems (Confluence, Jira, internal documentation) to find comprehensive answers to questions about internal concepts, systems, and terminology. This skill performs parallel searches across multiple sources and synthesizes results with proper citations. + +**Use this skill when:** Users ask about internal company knowledge that might be documented in Confluence pages, Jira tickets, or internal documentation. + +--- + +## Workflow + +Follow this 5-step process to provide comprehensive, well-cited answers: + +### Step 1: Identify Search Query + +Extract the core search terms from the user's question. + +**Examples:** +- User: "Find everything about Stratus minions" → Search: "Stratus minions" +- User: "What do we know about the billing system?" → Search: "billing system" +- User: "Explain our deployment process" → Search: "deployment process" + +**Consider:** +- Main topic or concept +- Any specific system/component names +- Technical terms or jargon + +--- + +### Step 2: Execute Parallel Search + +Search across all available knowledge sources simultaneously for comprehensive coverage. + +#### Option A: Cross-System Search (Recommended First) + +Use the **`search`** tool (Rovo Search) to search across Confluence and Jira at once: + +``` +search( + cloudId="...", + query="[extracted search terms]" +) +``` + +**When to use:** +- Default approach for most queries +- When you don't know which system has the information +- Fastest way to get results from multiple sources + +**Example:** +``` +search( + cloudId="...", + query="Stratus minions" +) +``` + +This returns results from both Confluence pages and Jira issues. + +#### Option B: Targeted Confluence Search + +Use **`searchConfluenceUsingCql`** when specifically searching Confluence: + +``` +searchConfluenceUsingCql( + cloudId="...", + cql="text ~ 'search terms' OR title ~ 'search terms'" +) +``` + +**When to use:** +- User specifically mentions "in Confluence" or "in our docs" +- Cross-system search returns too many Jira results +- Looking for documentation rather than tickets + +**Example CQL patterns:** +``` +text ~ "Stratus minions" +text ~ "authentication" AND type = page +title ~ "deployment guide" +``` + +#### Option C: Targeted Jira Search + +Use **`searchJiraIssuesUsingJql`** when specifically searching Jira: + +``` +searchJiraIssuesUsingJql( + cloudId="...", + jql="text ~ 'search terms' OR summary ~ 'search terms'" +) +``` + +**When to use:** +- User mentions "tickets", "issues", or "bugs" +- Looking for historical problems or implementation details +- Cross-system search returns mostly documentation + +**Example JQL patterns:** +``` +text ~ "Stratus minions" +summary ~ "authentication" AND type = Bug +text ~ "deployment" AND created >= -90d +``` + +#### Search Strategy + +**For most queries, use this sequence:** + +1. Start with `search` (cross-system) - **always try this first** +2. If results are unclear, follow up with targeted searches +3. If results mention specific pages/tickets, fetch them for details + +--- + +### Step 3: Fetch Detailed Content + +After identifying relevant sources, fetch full content for comprehensive answers. + +#### For Confluence Pages + +When search results reference Confluence pages: + +``` +getConfluencePage( + cloudId="...", + pageId="[page ID from search results]", + contentFormat="markdown" +) +``` + +**Returns:** Full page content in Markdown format + +**When to fetch:** +- Search result snippet is too brief +- Need complete context +- Page seems to be the primary documentation + +#### For Jira Issues + +When search results reference Jira issues: + +``` +getJiraIssue( + cloudId="...", + issueIdOrKey="PROJ-123" +) +``` + +**Returns:** Full issue details including description, comments, status + +**When to fetch:** +- Need to understand a reported bug or issue +- Search result doesn't show full context +- Issue contains important implementation notes + +#### Prioritization + +**Fetch in this order:** +1. **Official documentation pages** (Confluence pages with "guide", "documentation", "overview" in title) +2. **Recent/relevant issues** (Jira tickets that are relevant and recent) +3. **Additional context** (related pages mentioned in initial results) + +**Don't fetch everything** - be selective based on relevance to user's question. + +--- + +### Step 4: Synthesize Results + +Combine information from multiple sources into a coherent answer. + +#### Synthesis Guidelines + +**Structure your answer:** + +1. **Direct Answer First** + - Start with a clear, concise answer to the question + - "Stratus minions are..." + +2. **Detailed Explanation** + - Provide comprehensive details from all sources + - Organize by topic, not by source + +3. **Source Attribution** + - Note where each piece of information comes from + - Format: "According to [source], ..." + +4. **Highlight Discrepancies** + - If sources conflict, note it explicitly + - Example: "The Confluence documentation states X, however Jira ticket PROJ-123 indicates that due to bug Y, the behavior is actually Z" + +5. **Provide Context** + - Mention if information is outdated + - Note if a feature is deprecated or in development + +#### Synthesis Patterns + +**Pattern 1: Multiple sources agree** +``` +Stratus minions are background worker processes that handle async tasks. + +According to the Confluence documentation, they process jobs from the queue and +can be scaled horizontally. This is confirmed by several Jira tickets (PROJ-145, +PROJ-203) which discuss minion configuration and scaling strategies. +``` + +**Pattern 2: Sources provide different aspects** +``` +The billing system has two main components: + +**Payment Processing** (from Confluence "Billing Architecture" page) +- Handles credit card transactions +- Integrates with Stripe API +- Runs nightly reconciliation + +**Invoice Generation** (from Jira PROJ-189) +- Creates monthly invoices +- Note: Currently has a bug where tax calculation fails for EU customers +- Fix planned for Q1 2024 +``` + +**Pattern 3: Conflicting information** +``` +There is conflicting information about the authentication timeout: + +- **Official Documentation** (Confluence) states: 30-minute session timeout +- **Implementation Reality** (Jira PROJ-456, filed Oct 2023): Actual timeout is + 15 minutes due to load balancer configuration +- **Status:** Engineering team aware, fix planned but no timeline yet + +Current behavior: Expect 15-minute timeout despite docs saying 30 minutes. +``` + +**Pattern 4: Incomplete information** +``` +Based on available documentation: + +[What we know about deployment process from Confluence and Jira] + +However, I couldn't find information about: +- Rollback procedures +- Database migration handling + +You may want to check with the DevOps team or search for additional documentation. +``` + +--- + +### Step 5: Provide Citations + +Always include links to source materials so users can explore further. + +#### Citation Format + +**For Confluence pages:** +``` +**Source:** [Page Title](https://yoursite.atlassian.net/wiki/spaces/SPACE/pages/123456) +``` + +**For Jira issues:** +``` +**Related Tickets:** +- [PROJ-123](https://yoursite.atlassian.net/browse/PROJ-123) - Brief description +- [PROJ-456](https://yoursite.atlassian.net/browse/PROJ-456) - Brief description +``` + +**Complete citation section:** +``` +## Sources + +**Confluence Documentation:** +- [Stratus Architecture Guide](https://yoursite.atlassian.net/wiki/spaces/DOCS/pages/12345) +- [Minion Configuration](https://yoursite.atlassian.net/wiki/spaces/DEVOPS/pages/67890) + +**Jira Issues:** +- [PROJ-145](https://yoursite.atlassian.net/browse/PROJ-145) - Minion scaling implementation +- [PROJ-203](https://yoursite.atlassian.net/browse/PROJ-203) - Performance optimization + +**Additional Resources:** +- [Internal architecture doc link if found] +``` + +--- + +## Search Best Practices + +### Effective Search Terms + +**Do:** +- ✅ Use specific technical terms: "OAuth authentication flow" +- ✅ Include system names: "Stratus minions" +- ✅ Use acronyms if they're common: "API rate limiting" +- ✅ Try variations if first search fails: "deploy process" → "deployment pipeline" + +**Don't:** +- ❌ Be too generic: "how things work" +- ❌ Use full sentences: Use key terms instead +- ❌ Include filler words: "the", "our", "about" + +### Search Result Quality + +**Good results:** +- Recent documentation (< 1 year old) +- Official/canonical pages (titled "Guide", "Documentation", "Overview") +- Multiple sources confirming same information +- Detailed implementation notes + +**Questionable results:** +- Very old tickets (> 2 years, may be outdated) +- Duplicate or conflicting information +- Draft pages or work-in-progress docs +- Personal pages (may not be official) + +**When results are poor:** +- Try different search terms +- Expand search to include related concepts +- Search for specific error messages or codes +- Ask user for more context + +--- + +## Handling Common Scenarios + +### Scenario 1: No Results Found + +If searches return no results: + +``` +I searched across Confluence and Jira but couldn't find information about "[topic]". + +This could mean: +- The concept hasn't been documented yet +- It might be called something else (can you provide alternative names?) +- Documentation might be in a different system I don't have access to + +Would you like me to: +1. Try searching with different terms? +2. Search for related concepts? +3. Look for mentions in recent Jira tickets? +``` + +### Scenario 2: Too Many Results + +If search returns 50+ results: + +- Focus on most recent and most relevant (top 10-15) +- Prioritize official documentation over random tickets +- Group similar results: "I found 15 tickets related to authentication bugs..." +- Offer to narrow search: "Would you like me to focus on [specific aspect]?" + +### Scenario 3: Ambiguous Query + +If the user's question is vague: + +``` +I found information about several topics related to "[term]": +1. [Topic A] - from Confluence page X +2. [Topic B] - from Jira tickets Y, Z +3. [Topic C] - from documentation W + +Which aspect are you most interested in? +``` + +### Scenario 4: Outdated Information + +If you find information is old: + +``` +I found documentation about [topic], but note that it's from [date] and may be outdated. + +[Provide the information anyway] + +You may want to verify this is still current, especially if it's from more than +a year ago. +``` + +### Scenario 5: Restricted/Private Content + +If search returns inaccessible content: + +``` +Search results mention [topic] in the following locations, but I don't have +access to view the full content: +- Confluence page: [title] (Space: [SPACE]) +- Jira issue: PROJ-123 + +You may need to access these directly or check your permissions. +``` + +--- + +## Examples + +### Example 1: Clear Documentation Available + +**User:** "What are Stratus minions?" + +**Process:** +1. Search: `search(cloudId="...", query="Stratus minions")` +2. Find: Confluence page "Stratus Architecture" + 3 Jira tickets +3. Fetch: Get full Confluence page content +4. Synthesize: Combine architectural overview with implementation details from tickets +5. Cite: Provide links to all sources + +**Answer:** +``` +Stratus minions are background worker processes that handle asynchronous task +processing in the Stratus system. + +**Key Characteristics:** +- Run independently from web servers +- Process jobs from a Redis queue +- Can be scaled horizontally for increased throughput +- Handle tasks like email sending, report generation, and data processing + +**Architecture:** +According to the Stratus Architecture documentation, minions follow a +worker-queue pattern where: +1. Web application enqueues tasks +2. Minions poll the queue for work +3. Tasks are processed asynchronously +4. Results are stored in the database + +**Implementation Notes:** +From Jira ticket PROJ-203, the current deployment runs 5 minion instances, +with plans to scale to 10 for handling increased load. + +## Sources +- [Stratus Architecture Guide](link) +- [PROJ-203](link) - Minion scaling implementation +``` + +### Example 2: Conflicting Information + +**User:** "How long is our session timeout?" + +**Process:** +1. Search across Confluence and Jira +2. Find: Docs say 30 min, but ticket reports actual timeout is 15 min +3. Fetch both sources for complete context +4. Clearly note the discrepancy + +**Answer:** +``` +There is a discrepancy between documented and actual behavior: + +**Official Documentation** (Security Guidelines): +Session timeout is configured for 30 minutes of inactivity. + +**Actual Behavior** (PROJ-456, reported Oct 2023): +Users are being logged out after 15 minutes due to a load balancer configuration +that overrides the application setting. + +**Current Status:** +The engineering team is aware of this issue. The load balancer timeout needs to +be adjusted to match the application setting. No fix timeline has been set. + +**Recommendation:** Plan for 15-minute timeout until this is resolved. + +## Sources +- [Security Guidelines](link) - Official documentation +- [PROJ-456](link) - Timeout discrepancy bug report +``` + +### Example 3: Partial Information + +**User:** "Explain our deployment process" + +**Process:** +1. Search finds some information but not complete +2. Synthesize what is available +3. Note gaps in documentation + +**Answer:** +``` +Based on available documentation, here's what I found about the deployment process: + +**Build Phase** (from Confluence "CI/CD Guide"): +1. Code pushed to main branch triggers GitHub Actions +2. Automated tests run (unit, integration) +3. Docker image built and pushed to registry + +**Deployment Phase** (from Jira PROJ-789): +4. Staging deployment happens automatically +5. Manual approval required for production +6. Blue-green deployment strategy used + +**Gaps in Documentation:** +I couldn't find documented information about: +- Rollback procedures +- Database migration handling +- Post-deployment verification steps + +These details may exist in tribal knowledge or need to be documented. + +## Sources +- [CI/CD Guide](link) +- [PROJ-789](link) - Deployment pipeline implementation + +Would you like me to search for more specific aspects of deployment? +``` + +--- + +## Tips for High-Quality Answers + +### Do: +✅ Always search before answering (don't rely on general knowledge) +✅ Cite all sources with links +✅ Note discrepancies explicitly +✅ Mention when information is old +✅ Provide context and examples +✅ Structure answers clearly with headers +✅ Link to related documentation + +### Don't: +❌ Assume general knowledge applies to this company +❌ Make up information if search returns nothing +❌ Ignore conflicting information +❌ Quote entire documents (summarize instead) +❌ Overwhelm with too many sources (curate top 5-10) +❌ Forget to fetch details when snippets are insufficient + +--- + +## When NOT to Use This Skill + +This skill is for **internal company knowledge only**. Do NOT use for: + +❌ General technology questions (use your training knowledge) +❌ External documentation (use web_search) +❌ Company-agnostic questions +❌ Questions about other companies +❌ Current events or news + +**Examples of what NOT to use this skill for:** +- "What is machine learning?" (general knowledge) +- "How does React work?" (external documentation) +- "What's the weather?" (not knowledge search) +- "Find a restaurant" (not work-related) + +--- + +## Quick Reference + +**Primary tool:** `search(cloudId, query)` - Use this first, always + +**Follow-up tools:** +- `getConfluencePage(cloudId, pageId, contentFormat)` - Get full page content +- `getJiraIssue(cloudId, issueIdOrKey)` - Get full issue details +- `searchConfluenceUsingCql(cloudId, cql)` - Targeted Confluence search +- `searchJiraIssuesUsingJql(cloudId, jql)` - Targeted Jira search + +**Answer structure:** +1. Direct answer +2. Detailed explanation +3. Source attribution +4. Discrepancies (if any) +5. Citations with links + +**Remember:** +- Parallel search > Sequential search +- Synthesize, don't just list +- Always cite sources +- Note conflicts explicitly +- Be clear about gaps in documentation diff --git a/.github/skills/superpowers-brainstorming/SKILL.md b/.github/skills/superpowers-brainstorming/SKILL.md index 47afc6e5..41c68c6f 100644 --- a/.github/skills/superpowers-brainstorming/SKILL.md +++ b/.github/skills/superpowers-brainstorming/SKILL.md @@ -183,3 +183,18 @@ A question about a UI topic is not automatically a visual question. "What does p If they agree to the companion, read the detailed guide before proceeding: `skills/brainstorming/visual-companion.md` + + +## Local guided-question contract + +This repository-owned contract overrides any earlier instruction to ask one question at a time. + +- Ask all currently known questions in numbered bulk question blocks. +- Use `Question`, `Recommendation`, `Why`, and `Default if accepted` for every + numbered question. +- Make `Recommendation` the suggested answer and `Why` its concrete rationale. +- Keep each question, recommendation, and reason brief, clear, and + decision-ready. +- Put unresolved follow-ups in another numbered block. If only one blocking + question remains, present it as a numbered one-item block. + diff --git a/.github/skills/superpowers-brainstorming/visual-companion.md b/.github/skills/superpowers-brainstorming/visual-companion.md index 7b89f6b2..906c9ac8 100644 --- a/.github/skills/superpowers-brainstorming/visual-companion.md +++ b/.github/skills/superpowers-brainstorming/visual-companion.md @@ -74,6 +74,13 @@ On Windows, the script auto-detects and switches to foreground mode (which block scripts/start-server.sh --project-dir /path/to/project --open ``` +**Gemini CLI:** +```bash +# Use --foreground and set is_background: true on your shell tool call +# so the process survives across turns +scripts/start-server.sh --project-dir /path/to/project --open --foreground +``` + **Copilot CLI:** ```bash # Use --foreground and start the server via the bash tool with mode: "async" diff --git a/.github/skills/superpowers-dispatching-parallel-agents/SKILL.md b/.github/skills/superpowers-dispatching-parallel-agents/SKILL.md index 7cf45c3b..602e727b 100644 --- a/.github/skills/superpowers-dispatching-parallel-agents/SKILL.md +++ b/.github/skills/superpowers-dispatching-parallel-agents/SKILL.md @@ -158,15 +158,6 @@ Agent 3 → Fix tool-approval-race-conditions.test.ts **Integration:** All fixes independent, no conflicts, full suite green -**Time saved:** 3 problems solved in parallel vs sequentially - -## Key Benefits - -1. **Parallelization** - Multiple investigations happen simultaneously -2. **Focus** - Each agent has narrow scope, less context to track -3. **Independence** - Agents don't interfere with each other -4. **Speed** - 3 problems solved in time of 1 - ## Verification After agents return: @@ -174,12 +165,3 @@ After agents return: 2. **Check for conflicts** - Did agents edit same code? 3. **Run full suite** - Verify all fixes work together 4. **Spot check** - Agents can make systematic errors - -## Real-World Impact - -From debugging session (2025-10-03): -- 6 failures across 3 files -- 3 agents dispatched in parallel -- All investigations completed concurrently -- All fixes integrated successfully -- Zero conflicts between agent changes diff --git a/.github/skills/superpowers-executing-plans/SKILL.md b/.github/skills/superpowers-executing-plans/SKILL.md index c8c569e9..2c3d4c24 100644 --- a/.github/skills/superpowers-executing-plans/SKILL.md +++ b/.github/skills/superpowers-executing-plans/SKILL.md @@ -11,15 +11,16 @@ Load plan, review critically, execute all tasks, report when complete. **Announce at start:** "I'm using the executing-plans skill to implement this plan." -**Note:** Tell your human partner that Superpowers works much better with access to subagents. The quality of its work will be significantly higher if run on a platform with subagent support (Claude Code, Codex CLI, Codex App, and Copilot CLI all qualify; see the per-platform tool refs in `../using-superpowers/references/`). If subagents are available, use superpowers-subagent-driven-development instead of this skill. +**Note:** Tell your human partner that Superpowers works much better with access to subagents (Claude Code, Codex CLI, Codex App, Copilot CLI, and Gemini CLI all qualify; see the per-platform tool refs in `../using-superpowers/references/`). If subagents are available, use superpowers-subagent-driven-development instead of this skill. ## The Process ### Step 1: Load and Review Plan -1. Read plan file -2. Review critically - identify any questions or concerns about the plan -3. If concerns: Raise them with your human partner before starting -4. If no concerns: Create todos for the plan items and proceed +1. Ensure an isolated workspace: use superpowers-using-git-worktrees to create one or verify the existing one +2. Read plan file +3. Review critically - identify any questions or concerns about the plan +4. If concerns: Raise them with your human partner before starting +5. If no concerns: Create todos for the plan items and proceed ### Step 2: Execute Tasks @@ -61,10 +62,3 @@ After all tasks complete and verified: - Reference skills when plan says to - Stop when blocked, don't guess - Never start implementation on main/master branch without explicit user consent - -## Integration - -**Required workflow skills:** -- **superpowers-using-git-worktrees** - Ensures isolated workspace (creates one or verifies existing) -- **superpowers-writing-plans** - Creates the plan this skill executes -- **superpowers-finishing-a-development-branch** - Complete development after all tasks diff --git a/.github/skills/superpowers-finishing-a-development-branch/SKILL.md b/.github/skills/superpowers-finishing-a-development-branch/SKILL.md index 0f8134bd..ccbf3f15 100644 --- a/.github/skills/superpowers-finishing-a-development-branch/SKILL.md +++ b/.github/skills/superpowers-finishing-a-development-branch/SKILL.md @@ -1,71 +1,58 @@ --- name: superpowers-finishing-a-development-branch -description: Use when implementation is complete, all tests pass, and you need to decide how to integrate the work - guides completion of development work by presenting structured options for merge, PR, or cleanup +description: Use when implementation is complete, all tests pass, and you need to decide how to integrate the work --- # Finishing a Development Branch ## Overview -Guide completion of development work by presenting clear options and handling chosen workflow. - **Core principle:** Verify tests → Detect environment → Present options → Execute choice → Clean up. **Announce at start:** "I'm using the finishing-a-development-branch skill to complete this work." -## The Process - -### Step 1: Verify Tests +## Step 1: Verify Tests -**Before presenting options, verify tests pass:** +Run the project's full test suite (`npm test` / `cargo test` / `pytest` / `go test ./...`). -```bash -# Run project's test suite -npm test / cargo test / pytest / go test ./... -``` +**If tests fail**, report the failures and stop — the menu comes after a green suite: -**If tests fail:** ``` Tests failing ( failures). Must fix before completing: [Show failures] - -Cannot proceed with merge/PR until tests pass. ``` -Stop. Don't proceed to Step 2. - -**If tests pass:** Continue to Step 2. +**If tests pass:** continue to Step 2. -### Step 2: Detect Environment - -**Determine workspace state before presenting options:** +## Step 2: Detect Environment ```bash GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P) GIT_COMMON=$(cd "$(git rev-parse --git-common-dir)" 2>/dev/null && pwd -P) +# Capture now, while still inside the workspace — Step 5 changes directory +# before cleanup (Step 6) needs this value +WORKTREE_PATH=$(git rev-parse --show-toplevel) ``` This determines which menu to show and how cleanup works: | State | Menu | Cleanup | |-------|------|---------| -| `GIT_DIR == GIT_COMMON` (normal repo) | Standard 4 options | No worktree to clean up | -| `GIT_DIR != GIT_COMMON`, named branch | Standard 4 options | Provenance-based (see Step 6) | -| `GIT_DIR != GIT_COMMON`, detached HEAD | Reduced 3 options (no merge) | No cleanup (externally managed) | - -### Step 3: Determine Base Branch +| `GIT_DIR == GIT_COMMON` (normal repo) | Standard 3 options | No worktree to clean up | +| `GIT_DIR != GIT_COMMON`, named branch | Standard 3 options | Provenance-based (see Step 6) | +| `GIT_DIR != GIT_COMMON`, detached HEAD | Reduced 2 options (no merge) | Externally managed — leave in place | -```bash -# Try common base branches -git merge-base HEAD main 2>/dev/null || git merge-base HEAD master 2>/dev/null -``` +## Step 3: Determine Base Branch -Or ask: "This branch split from main - is that correct?" +The base branch is whatever this work forked from — usually named in the +plan, the conversation, or the branch's upstream. If it is not already +known, ask: "This branch split from - is that correct?" +Confirm before merging: merging into the wrong base is expensive to undo. -### Step 4: Present Options +## Step 4: Present Options -**Normal repo and named-branch worktree — present exactly these 4 options:** +**Normal repo and named-branch worktree — present exactly these 3 options:** ``` Implementation complete. What would you like to do? @@ -73,28 +60,30 @@ Implementation complete. What would you like to do? 1. Merge back to locally 2. Push and create a Pull Request 3. Keep the branch as-is (I'll handle it later) -4. Discard this work Which option? ``` -**Detached HEAD — present exactly these 3 options:** +**Detached HEAD — present exactly these 2 options:** ``` Implementation complete. You're on a detached HEAD (externally managed workspace). 1. Push as new branch and create a Pull Request 2. Keep as-is (I'll handle it later) -3. Discard this work Which option? ``` -**Don't add explanation** - keep options concise. +Present the menu exactly as written — concise, with every option coming +from the list above. Discarding the work happens only in response to your +human partner explicitly asking for it (see "If your human partner asks to +discard the work" below). Wait for their answer; the integration decision +is theirs. -### Step 5: Execute Choice +## Step 5: Execute Choice -#### Option 1: Merge Locally +### Option 1: Merge Locally ```bash # Get main repo root for CWD safety @@ -108,34 +97,43 @@ git merge # Verify tests on merged result - -# Only after merge succeeds: cleanup worktree (Step 6), then delete branch ``` -Then: Cleanup worktree (Step 6), then delete branch: +If tests fail on the merged result: stop, leave the worktree and branch in +place, and investigate — nothing has been pushed, so the merge is local +and recoverable. + +Once the merged result is green: clean up the worktree (Step 6), then +delete the branch: ```bash git branch -d ``` -#### Option 2: Push and Create PR +### Option 2: Push and Create PR ```bash -# Push branch git push -u origin +# From a detached HEAD, name the new branch on the remote: +# git push origin HEAD:refs/heads/ ``` -**Do NOT clean up worktree** — user needs it alive to iterate on PR feedback. +Then create the pull/merge request against with the forge's +tooling — its CLI if one is available, or the creation URL most forges +print when you push — following the repo's PR template and conventions if +present, and report the URL to your human partner. + +Keep the worktree — your human partner iterates on PR feedback there. -#### Option 3: Keep As-Is +### Option 3: Keep As-Is Report: "Keeping branch . Worktree preserved at ." -**Don't cleanup worktree.** +### If your human partner asks to discard the work -#### Option 4: Discard +This path exists only as a response to an explicit request to throw the +work away. Confirm first: -**Confirm first:** ``` This will permanently delete: - Branch @@ -145,41 +143,39 @@ This will permanently delete: Type 'discard' to confirm. ``` -Wait for exact confirmation. +Wait for that exact confirmation. When it arrives: -If confirmed: ```bash MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel) cd "$MAIN_ROOT" ``` -Then: Cleanup worktree (Step 6), then force-delete branch: +Then clean up the worktree (Step 6) and force-delete the branch: + ```bash git branch -D ``` -### Step 6: Cleanup Workspace +## Step 6: Cleanup Workspace -**Only runs for Options 1 and 4.** Options 2 and 3 always preserve the worktree. - -```bash -GIT_DIR=$(cd "$(git rev-parse --git-dir)" 2>/dev/null && pwd -P) -GIT_COMMON=$(cd "$(git rev-parse --git-common-dir)" 2>/dev/null && pwd -P) -WORKTREE_PATH=$(git rev-parse --show-toplevel) -``` +**Runs for Option 1 and confirmed discards.** Options 2 and 3 always +preserve the worktree. Both callers have already changed directory to the +main repo root — worktree removal must run from outside the worktree — +and use the `GIT_DIR`/`GIT_COMMON`/`WORKTREE_PATH` values captured in +Step 2, from before that directory change. **If `GIT_DIR == GIT_COMMON`:** Normal repo, no worktree to clean up. Done. -**If worktree path is under `.worktrees/` or `worktrees/`:** Superpowers created this worktree — we own cleanup. +**If `WORKTREE_PATH` is under `.worktrees/` or `worktrees/`:** Superpowers +created this worktree — we own cleanup: ```bash -MAIN_ROOT=$(git -C "$(git rev-parse --git-common-dir)/.." rev-parse --show-toplevel) -cd "$MAIN_ROOT" git worktree remove "$WORKTREE_PATH" git worktree prune # Self-healing: clean up any stale registrations ``` -**Otherwise:** The host environment (harness) owns this workspace. Do NOT remove it. If your platform provides a workspace-exit tool, use it. Otherwise, leave the workspace in place. +**Otherwise:** The host environment owns this workspace — leave it in +place. If your platform provides a workspace-exit tool, use it. ## Quick Reference @@ -188,54 +184,18 @@ git worktree prune # Self-healing: clean up any stale registrations | 1. Merge locally | yes | - | - | yes | | 2. Create PR | - | yes | yes | - | | 3. Keep as-is | - | - | yes | - | -| 4. Discard | - | - | - | yes (force) | - -## Common Mistakes - -**Skipping test verification** -- **Problem:** Merge broken code, create failing PR -- **Fix:** Always verify tests before offering options - -**Open-ended questions** -- **Problem:** "What should I do next?" is ambiguous -- **Fix:** Present exactly 4 structured options (or 3 for detached HEAD) - -**Cleaning up worktree for Option 2** -- **Problem:** Remove worktree user needs for PR iteration -- **Fix:** Only cleanup for Options 1 and 4 - -**Deleting branch before removing worktree** -- **Problem:** `git branch -d` fails because worktree still references the branch -- **Fix:** Merge first, remove worktree, then delete branch - -**Running git worktree remove from inside the worktree** -- **Problem:** Command fails silently when CWD is inside the worktree being removed -- **Fix:** Always `cd` to main repo root before `git worktree remove` - -**Cleaning up harness-owned worktrees** -- **Problem:** Removing a worktree the harness created causes phantom state -- **Fix:** Only clean up worktrees under `.worktrees/` or `worktrees/` - -**No confirmation for discard** -- **Problem:** Accidentally delete work -- **Fix:** Require typed "discard" confirmation - -## Red Flags - -**Never:** -- Proceed with failing tests -- Merge without verifying tests on result -- Delete work without confirmation -- Force-push without explicit request -- Remove a worktree before confirming merge success -- Clean up worktrees you didn't create (provenance check) -- Run `git worktree remove` from inside the worktree - -**Always:** -- Verify tests before offering options -- Detect environment before presenting menu -- Present exactly 4 options (or 3 for detached HEAD) -- Get typed confirmation for Option 4 -- Clean up worktree for Options 1 & 4 only -- `cd` to main repo root before worktree removal -- Run `git worktree prune` after removal +| Discard (explicit request only) | - | - | - | yes (force) | + +## Common Rationalizations + +| Excuse | Reality | +|--------|---------| +| "Tests passed earlier this session" | Run the suite on the tree you are about to integrate. A green run only proves the tree it ran on. | +| "They obviously want it merged" | Integration is your human partner's decision. Present the menu and wait. | +| "They seem done with this feature — I'll offer to discard it" | The menu is complete as written. Discard happens only when your human partner asks for it in so many words. | +| "'Yeah, get rid of it' counts as confirmation" | Only the typed word `discard` authorizes deletion. | +| "The PR is up, so the worktree is clutter now" | PR feedback gets fixed in that worktree. It stays until the work lands. | +| "This other worktree looks stale — I'll clean it too" | Clean up only worktrees under `.worktrees/` or `worktrees/`. Everything else belongs to the host. | +| "The merged-result failure is probably flaky" | A failing merged result stops everything. Branch and worktree stay put while you investigate. | +| "The base branch is obviously main" | Confirm the fork point or ask. Merging into the wrong base is expensive to undo. | +| "The push was rejected — force-push will fix it" | A rejected push means the remote moved. Investigate; force-push only on your human partner's explicit request. | diff --git a/.github/skills/superpowers-receiving-code-review/SKILL.md b/.github/skills/superpowers-receiving-code-review/SKILL.md index a682d065..d78706d6 100644 --- a/.github/skills/superpowers-receiving-code-review/SKILL.md +++ b/.github/skills/superpowers-receiving-code-review/SKILL.md @@ -203,11 +203,3 @@ You understand 1,2,3,6. Unclear on 4,5. ## GitHub Thread Replies When replying to inline review comments on GitHub, reply in the comment thread (`gh api repos/{owner}/{repo}/pulls/{pr}/comments/{id}/replies`), not as a top-level PR comment. - -## The Bottom Line - -**External feedback = suggestions to evaluate, not orders to follow.** - -Verify. Question. Then implement. - -No performative agreement. Technical rigor always. diff --git a/.github/skills/superpowers-requesting-code-review/SKILL.md b/.github/skills/superpowers-requesting-code-review/SKILL.md index 01cacd48..592fa2a7 100644 --- a/.github/skills/superpowers-requesting-code-review/SKILL.md +++ b/.github/skills/superpowers-requesting-code-review/SKILL.md @@ -5,7 +5,7 @@ description: Use when completing tasks, implementing major features, or before m # Requesting Code Review -Dispatch a code reviewer subagent to catch issues before they cascade. The reviewer gets precisely crafted context for evaluation — never your session's history. This keeps the reviewer focused on the work product, not your thought process, and preserves your own context for continued work. +Dispatch a code reviewer subagent to catch issues before they cascade. The reviewer gets precisely crafted context for evaluation — never your session's history. **Core principle:** Review early, review often. @@ -72,20 +72,12 @@ You: [Fix progress indicators] [Continue to Task 3] ``` -## Integration with Workflows +## Common Rationalizations -**Subagent-Driven Development:** -- Review after EACH task -- Catch issues before they compound -- Fix before moving to next task - -**Executing Plans:** -- Review after each task or at natural checkpoints -- Get feedback, apply, continue - -**Ad-Hoc Development:** -- Review before merge -- Review when stuck +| Excuse | Reality | +|--------|---------| +| "I'll just review the diff myself instead of dispatching a reviewer" | You're the coordinator — reviewing the diff inline burns the context window you need to keep driving the work. Dispatch a reviewer subagent: the diff and the evaluation live in its context, and only the findings come back to you. | +| "The reviewer needs my whole session history to understand the change" | Hand it precisely crafted context, never your session's history. That keeps the reviewer on the work product, not your thought process. | ## Red Flags diff --git a/.github/skills/superpowers-subagent-driven-development/SKILL.md b/.github/skills/superpowers-subagent-driven-development/SKILL.md index e400a5a9..5f846003 100644 --- a/.github/skills/superpowers-subagent-driven-development/SKILL.md +++ b/.github/skills/superpowers-subagent-driven-development/SKILL.md @@ -51,38 +51,96 @@ digraph process { subgraph cluster_per_task { label="Per Task"; "Dispatch implementer subagent (./implementer-prompt.md)" [shape=box]; - "Implementer subagent asks questions?" [shape=diamond]; + "Implementer asks questions?" [shape=diamond]; "Answer questions, provide context" [shape=box]; - "Implementer subagent implements, tests, commits, self-reviews" [shape=box]; - "Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)" [shape=box]; - "Task reviewer reports spec ✅ and quality approved?" [shape=diamond]; - "Dispatch fix subagent for Critical/Important findings" [shape=box]; - "Mark task complete in todo list and progress ledger" [shape=box]; + "Implementer implements, tests, commits, self-reviews" [shape=box]; + "Generate review package, dispatch task reviewer (./task-reviewer-prompt.md)" [shape=box]; + "Spec ✅ and quality approved?" [shape=diamond]; + "Finding conflicts with plan text?" [shape=diamond]; + "Ask human partner which governs" [shape=box]; + "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [shape=box]; + "Dispatch scoped re-review (./re-review-prompt.md)" [shape=box]; + "All findings addressed?" [shape=diamond]; + "R = 5?" [shape=diamond]; + "Adjudicate each open finding" [shape=box]; + "Any load-bearing finding?" [shape=diamond]; + "STOP: report BLOCKED to human partner" [shape=box]; + "Park findings in ledger with rulings" [shape=box]; + "Append completion to ledger, mark todo complete" [shape=box]; } - "Read plan, note context and global constraints, create todos" [shape=box]; + "Setup: worktree, ledger check, read plan, pre-flight review" [shape=box]; "More tasks remain?" [shape=diamond]; - "Dispatch final code reviewer subagent (../requesting-code-review/code-reviewer.md)" [shape=box]; + "Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" [shape=box]; + "Final findings? ONE fix dispatch, one scoped re-review, adjudicate residuals" [shape=box]; + "Final review clean: delete this plan's workspace" [shape=box]; "Use superpowers-finishing-a-development-branch" [shape=box style=filled fillcolor=lightgreen]; - "Read plan, note context and global constraints, create todos" -> "Dispatch implementer subagent (./implementer-prompt.md)"; - "Dispatch implementer subagent (./implementer-prompt.md)" -> "Implementer subagent asks questions?"; - "Implementer subagent asks questions?" -> "Answer questions, provide context" [label="yes"]; - "Answer questions, provide context" -> "Dispatch implementer subagent (./implementer-prompt.md)"; - "Implementer subagent asks questions?" -> "Implementer subagent implements, tests, commits, self-reviews" [label="no"]; - "Implementer subagent implements, tests, commits, self-reviews" -> "Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)"; - "Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)" -> "Task reviewer reports spec ✅ and quality approved?"; - "Task reviewer reports spec ✅ and quality approved?" -> "Dispatch fix subagent for Critical/Important findings" [label="no"]; - "Dispatch fix subagent for Critical/Important findings" -> "Write diff file, dispatch task reviewer subagent (./task-reviewer-prompt.md)" [label="re-review"]; - "Task reviewer reports spec ✅ and quality approved?" -> "Mark task complete in todo list and progress ledger" [label="yes"]; - "Mark task complete in todo list and progress ledger" -> "More tasks remain?"; + "Setup: worktree, ledger check, read plan, pre-flight review" -> "Dispatch implementer subagent (./implementer-prompt.md)"; + "Dispatch implementer subagent (./implementer-prompt.md)" -> "Implementer asks questions?"; + "Implementer asks questions?" -> "Answer questions, provide context" [label="yes"]; + "Answer questions, provide context" -> "Implementer implements, tests, commits, self-reviews"; + "Implementer asks questions?" -> "Implementer implements, tests, commits, self-reviews" [label="no"]; + "Implementer implements, tests, commits, self-reviews" -> "Generate review package, dispatch task reviewer (./task-reviewer-prompt.md)"; + "Generate review package, dispatch task reviewer (./task-reviewer-prompt.md)" -> "Spec ✅ and quality approved?"; + "Spec ✅ and quality approved?" -> "Append completion to ledger, mark todo complete" [label="yes"]; + "Spec ✅ and quality approved?" -> "Finding conflicts with plan text?" [label="no"]; + "Finding conflicts with plan text?" -> "Ask human partner which governs" [label="yes"]; + "Ask human partner which governs" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model"; + "Finding conflicts with plan text?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no"]; + "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" -> "Dispatch scoped re-review (./re-review-prompt.md)"; + "Dispatch scoped re-review (./re-review-prompt.md)" -> "All findings addressed?"; + "All findings addressed?" -> "Append completion to ledger, mark todo complete" [label="yes"]; + "All findings addressed?" -> "R = 5?" [label="no"]; + "R = 5?" -> "Fix round R of 5: R≤3 resume implementer; R≥4 fresh implementer, more capable model" [label="no - next round"]; + "R = 5?" -> "Adjudicate each open finding" [label="yes - breaker trips"]; + "Adjudicate each open finding" -> "Any load-bearing finding?"; + "Any load-bearing finding?" -> "STOP: report BLOCKED to human partner" [label="yes"]; + "Any load-bearing finding?" -> "Park findings in ledger with rulings" [label="no"]; + "Park findings in ledger with rulings" -> "Append completion to ledger, mark todo complete"; + "Append completion to ledger, mark todo complete" -> "More tasks remain?"; "More tasks remain?" -> "Dispatch implementer subagent (./implementer-prompt.md)" [label="yes"]; - "More tasks remain?" -> "Dispatch final code reviewer subagent (../requesting-code-review/code-reviewer.md)" [label="no"]; - "Dispatch final code reviewer subagent (../requesting-code-review/code-reviewer.md)" -> "Use superpowers-finishing-a-development-branch"; + "More tasks remain?" -> "Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" [label="no"]; + "Dispatch final code reviewer (../requesting-code-review/code-reviewer.md)" -> "Final findings? ONE fix dispatch, one scoped re-review, adjudicate residuals"; + "Final findings? ONE fix dispatch, one scoped re-review, adjudicate residuals" -> "Final review clean: delete this plan's workspace"; + "Final review clean: delete this plan's workspace" -> "Use superpowers-finishing-a-development-branch"; } ``` -## Pre-Flight Plan Review +## Setup + +Ensure the work happens in an isolated workspace: use +superpowers-using-git-worktrees to create one or verify the existing one. +Never start implementation on a main/master branch without your human +partner's explicit consent. + +Conversation memory does not survive compaction. In real sessions, +controllers that lost their place have re-dispatched entire completed task +sequences — the single most expensive failure observed. Track progress in +a ledger file, not only in todos. + +- Each plan owns a workspace: at skill start, run this skill's + `scripts/sdd-workspace PLAN_FILE` — it prints the plan's git-ignored + directory (`/.superpowers/sdd//`), home to + every artifact for THIS plan: ledger, briefs, reports, review packages. + Another plan's directory is never yours to read or write. +- Check for this plan's ledger at `/progress.md`. If its first + line names your plan file, tasks with a `Task : complete` line are DONE + — do not re-dispatch them; resume at the first task without one. A task + whose last line is a fix round is mid-loop: resume the loop at the next + round. A ledger whose first line names a different plan file — or a stray + ledger at the old flat path `.superpowers/sdd/progress.md` — is another + plan's progress: leave it in place and start your own, fresh. +- Create the ledger with its identity as the first line: + `# SDD ledger — plan: `. +- The ledger is your recovery map: the commits it names exist in git even + when your context no longer remembers creating them. After compaction, + trust the ledger and `git log` over your own recollection. +- `git clean -fdx` will destroy the workspace (it's git-ignored scratch); if + that happens, recover from `git log`. + +Read the plan once, note its context and Global Constraints, and create a +todo per task. Before dispatching Task 1, scan the plan once for conflicts: @@ -110,7 +168,11 @@ capable available model, not the session default. **Review tasks**: choose the model with the same judgment, scaled to the diff's size, complexity, and risk. A small mechanical diff does not need the -most capable model; a subtle concurrency change does. +most capable model; a subtle concurrency change does. Scoped re-reviews of +small fix diffs take a cheap-to-mid tier. + +**Fix-loop escalation (rounds 4-5)**: use a model at least one tier above +the implementer that got stuck. **Always specify the model explicitly when dispatching a subagent.** An omitted model inherits your session's model — often the most capable and @@ -129,11 +191,51 @@ that implementer. Single-file mechanical fixes also take the cheapest tier. - Touches multiple files with integration concerns → standard model - Requires design judgment or broad codebase understanding → most capable model -## Handling Implementer Status +## The Task Loop + +Everything you paste into a dispatch prompt — and everything a subagent +prints back — stays resident in your context for the rest of the session +and is re-read on every later turn. Hand artifacts over as files. + +### 1. Dispatch the implementer + +Record BASE (`git rev-parse HEAD`) before dispatching — the review package +and fix-round diffs need it. + +- **Task brief:** before dispatching an implementer, run this skill's + `scripts/task-brief PLAN_FILE N` — it extracts the task's full text to a + uniquely named file and prints the path. Compose the dispatch so the + brief stays the single source of + requirements. Your dispatch should contain: (1) one line on where this + task fits in the project; (2) the brief path, introduced as "read this + first — it is your requirements, with the exact values to use verbatim"; + (3) interfaces and decisions from earlier tasks that the brief cannot + know; (4) your resolution of any ambiguity you noticed in the brief; + (5) the report-file path and report contract. Exact values (numbers, + magic strings, signatures, test cases) appear only in the brief. Never + make a subagent read the whole plan file. +- **Report file:** name the implementer's report file after the brief + (brief `…/task-N-brief.md` → report `…/task-N-report.md`) and put it in + the dispatch prompt. The implementer writes the full report there and + returns only status, commits, a one-line test summary, and concerns. +- A dispatch prompt describes one task, not the session's history. Do not + paste accumulated prior-task summaries ("state after Tasks 1-3") into + later dispatches — a real session's dispatch hit 42k chars of which 99% + was pasted history. A fresh subagent needs its task, the interfaces it + touches, and the global constraints. Nothing else. +- If an earlier task parked a finding in the area this task touches, carry + a pointer to that ledger entry in the dispatch. +- Record the implementer's agent identity from the dispatch result — + fix-loop rounds 1-3 resume this agent. +- Never dispatch multiple implementation subagents in parallel (conflicts). + +Template: [implementer-prompt.md](implementer-prompt.md) + +### 2. Handle the report Implementer subagents report one of four statuses. Handle each appropriately: -**DONE:** Generate the review package (`scripts/review-package BASE HEAD`, from this skill's directory — it prints the unique file path it wrote; BASE is the commit you recorded before dispatching the implementer — never `HEAD~1`, which silently drops all but the last commit of a multi-commit task), then dispatch the task reviewer with the printed path. +**DONE:** Generate the review package (`scripts/review-package PLAN_FILE BASE HEAD`, from this skill's directory — it prints the unique file path it wrote; BASE is the commit you recorded before dispatching the implementer — never `HEAD~1`, which silently drops all but the last commit of a multi-commit task), then dispatch the task reviewer with the printed path. **DONE_WITH_CONCERNS:** The implementer completed the work but flagged doubts. Read the concerns before proceeding. If the concerns are about correctness or scope, address them before review. If they're observations (e.g., "this file is getting large"), note them and proceed to review. @@ -147,20 +249,37 @@ Implementer subagents report one of four statuses. Handle each appropriately: **Never** ignore an escalation or force the same model to retry without changes. If the implementer said it's stuck, something needs to change. -## Handling Reviewer ⚠️ Items - -The task reviewer may report "⚠️ Cannot verify from diff" items — requirements -that live in unchanged code or span tasks. These do not block the rest of the -review, but you must resolve each one yourself before marking the task -complete: you hold the plan and cross-task context the reviewer -lacks. If you confirm an item is a real gap, treat it as a failed spec -review — send it back to the implementer and re-review. +If the implementer asks questions — before starting or mid-task — answer +clearly and completely, provide additional context if needed, and don't +rush it into implementation. -## Constructing Reviewer Prompts +### 3. Review the task Per-task reviews are task-scoped gates. The broad review happens once, at the -final whole-branch review. When you fill a reviewer template: +final whole-branch review. Never skip the task review, and never accept a +report missing either verdict — spec compliance AND task quality are both +required. Implementer self-review never replaces the task review; both are +needed. +- Hand the reviewer its diff as a file: run this skill's + `scripts/review-package PLAN_FILE BASE HEAD` and pass the reviewer the file path + it prints (or, without bash: `git log --oneline`, `git diff --stat`, + and `git diff -U10` for the range, redirected to one uniquely named + file). The output never enters your own context, and the reviewer sees + the commit list, stat summary, and full diff with context in one Read + call. Use the BASE you recorded before dispatching the implementer — + never `HEAD~1`, which silently truncates multi-commit tasks. Never + dispatch a task reviewer without a diff file. +- **Reviewer inputs:** the task reviewer gets three paths — the same brief + file, the report file, and the review package — plus the global + constraints that bind the task. +- The global-constraints block you hand the reviewer is its attention + lens. Copy the binding requirements verbatim from the plan's Global + Constraints section or the spec: exact values, exact formats, and the + stated relationships between components ("same layout as X", "matches + Y"). The reviewer's template already carries the process rules (YAGNI, + test hygiene, review method) — the constraints block is for what THIS + project's spec demands. - Do not add open-ended directives like "check all uses" or "run race tests if useful" without a concrete, task-specific reason - Do not ask a reviewer to re-run tests the implementer already ran on the @@ -171,110 +290,159 @@ final whole-branch review. When you fill a reviewer template: loop. If the prompt you are writing contains "do not flag," "don't treat X as a defect," "at most Minor," or "the plan chose" — stop: you are pre-judging, usually to spare yourself a review loop. -- The global-constraints block you hand the reviewer is its attention - lens. Copy the binding requirements verbatim from the plan's Global - Constraints section or the spec: exact values, exact formats, and the - stated relationships between components ("same layout as X", "matches - Y"). The reviewer's template already carries the process rules (YAGNI, - test hygiene, review method) — the constraints block is for what THIS - project's spec demands. -- Hand the reviewer its diff as a file: run this skill's - `scripts/review-package BASE HEAD` and pass the reviewer the file path - it prints (or, without bash: `git log --oneline`, `git diff --stat`, - and `git diff -U10` for the range, redirected to one uniquely named - file). The output never enters your own context, and the reviewer sees - the commit list, stat summary, and full diff with context in one Read - call. Use the BASE you recorded before dispatching the implementer — - never `HEAD~1`, which silently truncates multi-commit tasks. -- A dispatch prompt describes one task, not the session's history. Do not - paste accumulated prior-task summaries ("state after Tasks 1-3") into - later dispatches — a real session's dispatch hit 42k chars of which 99% - was pasted history. A fresh subagent needs its task, the interfaces it - touches, and the global constraints. Nothing else. -- Dispatch fix subagents for Critical and Important findings. Record Minor - findings in the progress ledger as you go, and point the final +The task reviewer may report "⚠️ Cannot verify from diff" items — requirements +that live in unchanged code or span tasks. These do not block the rest of the +review, but you must resolve each one yourself before marking the task +complete: you hold the plan and cross-task context the reviewer +lacks. If you confirm an item is a real gap, treat it as a failed spec +review — it enters the fix loop with the other findings. + +Template: [task-reviewer-prompt.md](task-reviewer-prompt.md) + +### 4. The fix loop + +The loop triggers when the review reports spec ❌, any Critical or Important +finding, or a ⚠️ item you confirmed as a real gap. + +Before the loop starts, two routes leave it immediately: + +- Record Minor findings in the progress ledger as you go + (`Task : minor (deferred): `), and point the final whole-branch review at that list so it can triage which must be fixed - before merge. A roll-up nobody reads is a silent discard. + before merge. A roll-up nobody reads is a silent discard. Minor findings + never enter the loop. - A finding labeled plan-mandated — or any finding that conflicts with what the plan's text requires — is the human's decision, like any plan contradiction: present the finding and the plan text, ask which governs. Do not dismiss the finding because the plan mandates it, and do not dispatch a fix that contradicts the plan without asking. -- The final whole-branch review gets a package too: run - `scripts/review-package MERGE_BASE HEAD` (MERGE_BASE = the commit the - branch started from, e.g. `git merge-base main HEAD`) and include the - printed path in the final review dispatch, so the final reviewer reads - one file instead of re-deriving the branch diff with git commands. -- Every fix dispatch carries the implementer contract: the fix subagent - re-runs the tests covering its change and reports the results. Name the - covering test files in the dispatch — a one-line fix does not need the - whole suite. Before re-dispatching the reviewer, confirm the fix report - contains the covering tests, the command run, and the output; dispatch - the re-review once all three are present. -- If the final whole-branch review returns findings, dispatch ONE fix - subagent with the complete findings list — not one fixer per finding. - Per-finding fixers each rebuild context and re-run suites; a real - session's final-review fix wave cost more than all its tasks combined. - -## File Handoffs - -Everything you paste into a dispatch prompt — and everything a subagent -prints back — stays resident in your context for the rest of the session -and is re-read on every later turn. Hand artifacts over as files: - -- **Task brief:** before dispatching an implementer, run this skill's - `scripts/task-brief PLAN_FILE N` — it extracts the task's full text to a - uniquely named file and prints the path. Compose the dispatch so the - brief stays the single source of requirements. Your dispatch should - contain: (1) one line on where this task fits in the project; (2) the - brief path, introduced as "read this first — it is your requirements, - with the exact values to use verbatim"; (3) interfaces and decisions - from earlier tasks that the brief cannot know; (4) your resolution of - any ambiguity you noticed in the brief; (5) the report-file path and - report contract. Exact values (numbers, magic strings, signatures, test - cases) appear only in the brief. -- **Report file:** name the implementer's report file after the brief - (brief `…/task-N-brief.md` → report `…/task-N-report.md`) and put it in - the dispatch prompt. The implementer writes the full report there and - returns only status, commits, a one-line test summary, and concerns. -- **Reviewer inputs:** the task reviewer gets three paths — the same brief - file, the report file, and the review package — plus the global - constraints that bind the task. -- Fix dispatches append their fix report (with test results) to the same - report file and return a short summary; re-reviews read the updated file. - -## Durable Progress - -Conversation memory does not survive compaction. In real sessions, -controllers that lost their place have re-dispatched entire completed task -sequences — the single most expensive failure observed. Track progress in -a ledger file, not only in todos. - -- At skill start, check for a ledger: - `cat "$(git rev-parse --show-toplevel)/.superpowers/sdd/progress.md"`. Tasks listed there - as complete are DONE — do not re-dispatch them; resume at the first task - not marked complete. -- When a task's review comes back clean, append one line to the ledger in - the same message as your other bookkeeping: - `Task N: complete (commits .., review clean)`. -- The ledger is your recovery map: the commits it names exist in git even - when your context no longer remembers creating them. After compaction, - trust the ledger and `git log` over your own recollection. -- `git clean -fdx` will destroy the ledger (it's git-ignored scratch); if - that happens, recover from `git log`. - -## Prompt Templates - -- [implementer-prompt.md](implementer-prompt.md) - Dispatch implementer subagent -- [task-reviewer-prompt.md](task-reviewer-prompt.md) - Dispatch task reviewer subagent (spec compliance + code quality) -- Final whole-branch review: use superpowers-requesting-code-review's [code-reviewer.md](../requesting-code-review/code-reviewer.md) +Everything else enters the loop. A fix round is one fix dispatch plus one +scoped re-review. Five rounds maximum per task: + +**Rounds 1-3 — resume the original implementer.** Send it the open findings +verbatim. Its context is intact: it knows the task, the code, and its own +choices. If your harness cannot send another message to a live subagent, +dispatch a fresh implementer carrying the brief path, the report-file path, +and the findings — the report file is the persistent memory either way. + +**Rounds 4-5 — dispatch a fresh implementer on a more capable model** (per +Model Selection), with the brief path, the report-file path, the open +findings, and this framing: "A prior implementer attempted this task +[N] times; you own it now. Read the report file for what was tried." A loop +that survives three resumes usually means the implementer cannot see its +own problem — fresh eyes and a capability bump in one move. + +**Every round, either way:** the implementer fixes, re-runs the tests +covering the amended code, appends its fix report to the same report file, +and returns the short contract. Before re-dispatching the reviewer, confirm +the fix report contains the covering tests, the command run, and the +output; dispatch the re-review once all three are present. Name the +covering test files in the fix message — a one-line fix does not need the +whole suite. + +**The re-review is scoped.** Run `scripts/review-package PLAN_FILE FIX_BASE HEAD` +where FIX_BASE is the head the previous review saw, and dispatch +[re-review-prompt.md](re-review-prompt.md) with the findings list, the +brief, the report file, and the printed diff path. The re-reviewer verdicts +each finding ADDRESSED or NOT ADDRESSED and flags new breakage in the fix +diff only. New Critical/Important breakage in the fix diff joins the open +findings list. Out-of-scope observations go to the ledger as deferred +minors — they never extend the loop. + +**After each round,** append to the ledger: +`Task : fix round /5 ( addressed, open — ; commits ..)` + +Never fix findings yourself in the controller session — your context stays +clean for coordination, and controller fixes skip review. + +**The breaker.** When round 5's re-review still leaves findings open, stop +dispatching. Adjudicate each open finding yourself — you hold the plan and +the cross-task context the reviewer lacks: + +- **The reviewer is wrong, or the point is contestable:** park it — + `Task : parked — — ruling: `. The final + review sees both sides. +- **Real, but nothing downstream builds on it:** park it the same way, with + a ruling that says it's real and deferred. +- **Real and load-bearing** — a later task builds on it, or it reveals a + plan defect: STOP. Append `Task : BLOCKED — ` and report to + your human partner with the finding, the plan text it collides with, and + the fix history. Parking a structural failure lets every dependent task + build on it and hands the final review a problem it cannot fix either. + +Adjudicate only at the cap. Adjudicating earlier to end a loop is +pre-judging with a different name. Every adjudication is a ledger entry — +a silent discard is forbidden. + +### 5. Complete the task + +When the review comes back clean — or every open finding is parked with a +ruling at the cap — append the completion line to the ledger in the same +message as your other bookkeeping: + +- `Task : complete (commits .., review clean)` +- `Task : complete (commits .., parked)` after a + tripped breaker + +Then mark the todo complete and move on. Never move to the next task while +the review has open Critical/Important issues that are neither fixed nor +parked-with-ruling at the cap. + +## Final Review + +The final whole-branch review gets a package too: run +`scripts/review-package PLAN_FILE MERGE_BASE HEAD` (MERGE_BASE = the commit the +branch started from, e.g. `git merge-base main HEAD`) and include the +printed path in the final review dispatch, so the final reviewer reads +one file instead of re-deriving the branch diff with git commands. Dispatch +on the most capable available model (see Model Selection), using +superpowers-requesting-code-review's +[code-reviewer.md](../requesting-code-review/code-reviewer.md). Point it at +the ledger's deferred-minor and parked lines so it can triage which must be +fixed before merge. + +If the final whole-branch review returns findings, dispatch ONE fix subagent +with the complete findings list — not one fixer per finding. +Per-finding fixers each rebuild context and re-run suites; a real +session's final-review fix wave cost more than all its tasks combined. +Then run exactly one scoped re-review of the fix wave +(`scripts/review-package PLAN_FILE FIX_BASE HEAD` over the fix range, +[re-review-prompt.md](re-review-prompt.md)). +Adjudicate any residual findings as in the task loop's breaker: park with +rulings, or stop on load-bearing ones. There is no second fix wave — +residual load-bearing findings surface to your human partner when +finishing-a-development-branch presents the options. + +## Finish + +When the final whole-branch review is clean and its fixes are merged, +delete this plan's workspace (`rm -rf `) — the git history is +the record now. Sibling directories belong to other plans; leave them +alone. + +Use superpowers-finishing-a-development-branch. + +## Common Rationalizations + +| Excuse | Reality | +|--------|---------| +| "Close enough on spec compliance" | Reviewer found spec gaps = not done. Fix or hit the cap and adjudicate — those are the only exits. | +| "I'll fix it myself, dispatching is overhead" | Controller fixes pollute your context and skip review. Resume the implementer. | +| "One more round will converge" | Past the cap, rounds don't converge — the failure is structural. Adjudicate and route. | +| "The reviewer will just find something new anyway" | Scoped re-reviews verify fixes; they cannot wander. New findings on untouched code go to the ledger, not the loop. | +| "This finding is obviously wrong, I'll drop it" | You adjudicate only at the cap, and every ruling is a ledger entry. Silent discards are forbidden. | +| "The fix was small, skip the re-review" | Unreviewed fixes are how regressions land. Every round ends with a scoped re-review. | +| "Reviews slow the loop down" | The loop without reviews is just unverified churn. Reviews are the loop's brakes and steering. | +| "Ledger bookkeeping is overhead" | The ledger is what survives compaction. Controllers without one have re-dispatched entire completed task sequences. | ## Example Workflow ``` You: I'm using Subagent-Driven Development to execute this plan. +[Setup: worktree verified] [Read plan file once: tmp/superpowers/plans/feature-plan.md] +[Resolve workspace: scripts/sdd-workspace tmp/superpowers/plans/feature-plan.md — no ledger inside, fresh start] [Create todos for all tasks] Task 1: Hook installation script @@ -285,134 +453,51 @@ Implementer: "Before I begin - should the hook be installed at user or system le You: "User level (~/.config/superpowers/hooks/)" -Implementer: "Got it. Implementing now..." -[Later] Implementer: +Implementer: [Later] - Implemented install-hook command - Added tests, 5/5 passing - Self-review: Found I missed --force flag, added it - Committed -[Run review-package, dispatch task reviewer with the printed path] +[Run review-package PLAN_FILE BASE HEAD; dispatch task reviewer with the printed path] Task reviewer: Spec ✅ - all requirements met, nothing extra. Strengths: Good test coverage, clean. Issues: None. Task quality: Approved. -[Mark Task 1 complete] +[Ledger: Task 1: complete (commits a1b2c3d..d4e5f6a, review clean)] Task 2: Recovery modes [Run task-brief for Task 2; dispatch implementer with brief + report paths + context] -Implementer: [No questions, proceeds] -Implementer: +Implementer: [No questions] - Added verify/repair modes - 8/8 tests passing - - Self-review: All good - Committed -[Run review-package, dispatch task reviewer with the printed path] +[Run review-package PLAN_FILE BASE HEAD; dispatch task reviewer with the printed path] Task reviewer: Spec ❌: - Missing: Progress reporting (spec says "report every 100 items") - - Extra: Added --json flag (not requested) Issues (Important): Magic number (100) -[Dispatch fix subagent with all findings] -Fixer: Removed --json flag, added progress reporting, extracted PROGRESS_INTERVAL constant +[Fix round 1: resume the implementer with both findings] +Implementer: Added progress reporting, extracted PROGRESS_INTERVAL constant. + Re-ran test/recovery.test.js — 10/10 passing. Fix report appended. -[Task reviewer reviews again] -Task reviewer: Spec ✅. Task quality: Approved. +[Run review-package PLAN_FILE FIX_BASE HEAD; dispatch scoped re-review] +Re-reviewer: Missing progress reporting — ADDRESSED (src/recovery.js:41). + Magic number — ADDRESSED (src/recovery.js:7). New breakage: none. + Verdict: all findings addressed. -[Mark Task 2 complete] +[Ledger: Task 2: fix round 1/5 (2 addressed, 0 open; commits d4e5f6a..b7c8d9e)] +[Ledger: Task 2: complete (commits d4e5f6a..b7c8d9e, review clean)] ... [After all tasks] -[Dispatch final code-reviewer] -Final reviewer: All requirements met, ready to merge +[Run review-package PLAN_FILE MERGE_BASE HEAD; dispatch final code-reviewer, most capable model] +Final reviewer: All requirements met. Deferred minors triaged: none block merge. -Done! -``` +[Delete this plan's workspace — the record now lives in git] -## Advantages - -**vs. Manual execution:** -- Subagents follow TDD naturally -- Fresh context per task (no confusion) -- Parallel-safe (subagents don't interfere) -- Subagent can ask questions (before AND during work) - -**vs. Executing Plans:** -- Same session (no handoff) -- Continuous progress (no waiting) -- Review checkpoints automatic - -**Efficiency gains:** -- Controller curates exactly what context is needed; bulk artifacts move - as files, not pasted text -- Subagent gets complete information upfront -- Questions surfaced before work begins (not after) - -**Quality gates:** -- Self-review catches issues before handoff -- Task review carries two verdicts: spec compliance and code quality -- Review loops ensure fixes actually work -- Spec compliance prevents over/under-building -- Code quality ensures implementation is well-built - -**Cost:** -- More subagent invocations (implementer + reviewer per task) -- Controller does more prep work (extracting all tasks upfront) -- Review loops add iterations -- But catches issues early (cheaper than debugging later) - -## Red Flags - -**Never:** -- Start implementation on main/master branch without explicit user consent -- Skip task review, or accept a report missing either verdict (spec compliance AND task quality are both required) -- Proceed with unfixed issues -- Dispatch multiple implementation subagents in parallel (conflicts) -- Make a subagent read the whole plan file (hand it its task brief — - `scripts/task-brief` — instead) -- Skip scene-setting context (subagent needs to understand where task fits) -- Ignore subagent questions (answer before letting them proceed) -- Accept "close enough" on spec compliance (reviewer found spec issues = not done) -- Skip review loops (reviewer found issues = implementer fixes = review again) -- Let implementer self-review replace actual review (both are needed) -- Tell a reviewer what not to flag, or pre-rate a finding's severity in the - dispatch prompt ("treat it as Minor at most") — the plan's example code is - a starting point, not evidence that its weaknesses were chosen -- Dispatch a task reviewer without a diff file — generate it first - (`scripts/review-package BASE HEAD`) and name the printed path in the - prompt -- Move to next task while the review has open Critical/Important issues -- Re-dispatch a task the progress ledger already marks complete — check - the ledger (and `git log`) after any compaction or resume - -**If subagent asks questions:** -- Answer clearly and completely -- Provide additional context if needed -- Don't rush them into implementation - -**If reviewer finds issues:** -- Implementer (same subagent) fixes them -- Reviewer reviews again -- Repeat until approved -- Don't skip the re-review - -**If subagent fails task:** -- Dispatch fix subagent with specific instructions -- Don't try to fix manually (context pollution) - -## Integration - -**Required workflow skills:** -- **superpowers-using-git-worktrees** - Ensures isolated workspace (creates one or verifies existing) -- **superpowers-writing-plans** - Creates the plan this skill executes -- **superpowers-requesting-code-review** - Code review template for the final whole-branch review -- **superpowers-finishing-a-development-branch** - Complete development after all tasks - -**Subagents should use:** -- **superpowers-test-driven-development** - Subagents follow TDD for each task - -**Alternative workflow:** -- **superpowers-executing-plans** - Use for parallel session instead of same-session execution +Done! Using superpowers-finishing-a-development-branch. +``` diff --git a/.github/skills/superpowers-subagent-driven-development/implementer-prompt.md b/.github/skills/superpowers-subagent-driven-development/implementer-prompt.md index 218fcfeb..fbe441e2 100644 --- a/.github/skills/superpowers-subagent-driven-development/implementer-prompt.md +++ b/.github/skills/superpowers-subagent-driven-development/implementer-prompt.md @@ -106,9 +106,12 @@ Subagent (general-purpose): ## After Review Findings - If a reviewer finds issues and you fix them, re-run the tests that cover - the amended code and append the results to your report file. Reviewers - will not re-run tests for you — your report is the test evidence. + If the task review finds issues, you will be resumed with the findings. + Fix them, re-run the tests that cover the amended code, and append a fix + report to your report file: what you changed, the covering tests you + ran, the command, and the output. Reviewers will not re-run tests for + you — your report is the test evidence. Then reply with the same short + status contract as your first report. ## Report Format diff --git a/.github/skills/superpowers-subagent-driven-development/re-review-prompt.md b/.github/skills/superpowers-subagent-driven-development/re-review-prompt.md new file mode 100644 index 00000000..18b0fb8a --- /dev/null +++ b/.github/skills/superpowers-subagent-driven-development/re-review-prompt.md @@ -0,0 +1,106 @@ +# Scoped Re-Review Prompt Template + +Use this template when dispatching a re-review after a fix round. The +re-reviewer verifies the findings were addressed and checks the fix diff for +new breakage. It is not a fresh review — the full review already happened. + +**Purpose:** Verify each finding from the previous review was addressed, and +that the fix itself broke nothing. + +``` +Subagent (general-purpose): + description: "Re-review Task N fix round R" + model: [MODEL — REQUIRED: choose per SKILL.md Model Selection; an omitted + model silently inherits the session's most expensive one] + prompt: | + You are re-reviewing one task's fix round. A previous review produced + findings; an implementer has attempted to fix them. Your job is to + verdict each finding and inspect the fix diff — nothing else. + + ## The Task + + Read the task brief: [BRIEF_FILE] + + ## The Findings Under Verification + + [FINDINGS] + + ## The Fix + + Read the implementer's report (fix reports are appended at the end): + [REPORT_FILE] + + **Fix base:** [FIX_BASE_SHA] (the head the previous review saw) + **Head:** [HEAD_SHA] + **Diff file:** [DIFF_FILE] + + Read the diff file once — it contains the fix commits, a stat summary, + and the fix diff with surrounding context. Do not re-run git commands. + If the diff file is missing, fetch the diff yourself: + `git diff --stat [FIX_BASE_SHA]..[HEAD_SHA]` and + `git diff [FIX_BASE_SHA]..[HEAD_SHA]`. + + Your review is read-only on this checkout. Do not mutate the working + tree, the index, HEAD, or branch state in any way. + + ## Scope + + Your scope is the findings list and the fix diff. Verdict every finding. + Inspect the fix diff for new problems the fix itself introduced. Do NOT + re-review code the fix did not touch: if you notice an issue entirely + outside the fix diff, report it under Out-of-Scope Observations — it + does not block this task and does not extend the loop. A broad + whole-branch review happens after all tasks are complete. + + ## Tests + + The implementer re-ran the tests covering the amended code and appended + the results to the report file. Treat the report as unverified claims: + confirm the fix report names the covering tests and shows their output, + and verify the claims against the diff. Do not re-run the suite to + confirm their report. Run a test only when reading the code raises a + specific doubt that no existing run answers — and then a focused test, + never a package-wide suite. + + ## Output Format + + Your final message is the report itself: begin directly with the first + finding's verdict. Every line is a verdict, a finding with file:line, + or a check you ran — no preamble, no process narration. + + ### Finding Verdicts + + For each finding in The Findings Under Verification, in order: + - **[finding one-liner]** — ADDRESSED | NOT ADDRESSED, with file:line + evidence. "Attempted" is not addressed: the specific defect must no + longer exist. + + ### New Breakage in the Fix Diff + + Anything the fix itself broke or introduced, with severity + (Critical/Important/Minor) and file:line. "None" if clean. + + ### Out-of-Scope Observations + + Issues you noticed entirely outside the fix diff. Non-blocking; the + controller ledgers these for the final review. "None" if none. + + ### Verdict + + **Fix round:** [All findings addressed, no new Critical/Important + breakage | Findings remain open] — list the open ones. +``` + +**Placeholders:** +- `[MODEL]` — REQUIRED: reviewer model per SKILL.md Model Selection; scoped + re-reviews of small fix diffs take a cheap-to-mid tier +- `[BRIEF_FILE]` — the task brief file (same file the implementer worked from) +- `[FINDINGS]` — the Critical/Important findings and spec gaps from the + previous review, copied verbatim, one per bullet +- `[REPORT_FILE]` — the implementer's report file (fix reports appended) +- `[FIX_BASE_SHA]` — the head the previous review saw +- `[HEAD_SHA]` — current commit +- `[DIFF_FILE]` — the path `scripts/review-package PLAN_FILE FIX_BASE HEAD` printed + +**Re-reviewer returns:** per-finding verdicts (ADDRESSED / NOT ADDRESSED), +new breakage in the fix diff, out-of-scope observations, and a round verdict. diff --git a/.github/skills/superpowers-subagent-driven-development/scripts/review-package b/.github/skills/superpowers-subagent-driven-development/scripts/review-package index 33bb20f7..31852e2a 100755 --- a/.github/skills/superpowers-subagent-driven-development/scripts/review-package +++ b/.github/skills/superpowers-subagent-driven-development/scripts/review-package @@ -4,26 +4,28 @@ # call. Using the recorded per-task BASE (not HEAD~1) keeps multi-commit # tasks intact. # -# Usage: review-package BASE HEAD [OUTFILE] -# Default OUTFILE: /.superpowers/sdd/review-...diff +# Usage: review-package PLAN_FILE BASE HEAD [OUTFILE] +# Default OUTFILE: /.superpowers/sdd//review-...diff # (named per range, so a re-review after fixes gets a distinct fresh file). set -euo pipefail -if [ $# -lt 2 ] || [ $# -gt 3 ]; then - echo "usage: review-package BASE HEAD [OUTFILE]" >&2 +if [ $# -lt 3 ] || [ $# -gt 4 ]; then + echo "usage: review-package PLAN_FILE BASE HEAD [OUTFILE]" >&2 exit 2 fi -base=$1 -head=$2 +plan=$1 +base=$2 +head=$3 +[ -f "$plan" ] || { echo "no such plan file: $plan" >&2; exit 2; } git rev-parse --verify --quiet "$base" >/dev/null || { echo "bad BASE: $base" >&2; exit 2; } git rev-parse --verify --quiet "$head" >/dev/null || { echo "bad HEAD: $head" >&2; exit 2; } -if [ $# -eq 3 ]; then - out=$3 +if [ $# -eq 4 ]; then + out=$4 else - dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace") + dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan") out="$dir/review-$(git rev-parse --short "$base")..$(git rev-parse --short "$head").diff" fi diff --git a/.github/skills/superpowers-subagent-driven-development/scripts/sdd-workspace b/.github/skills/superpowers-subagent-driven-development/scripts/sdd-workspace index ea9bb08f..4e2d1680 100755 --- a/.github/skills/superpowers-subagent-driven-development/scripts/sdd-workspace +++ b/.github/skills/superpowers-subagent-driven-development/scripts/sdd-workspace @@ -1,22 +1,40 @@ #!/usr/bin/env bash -# Resolve and ensure the working-tree directory SDD uses for its short-lived -# artifacts: task briefs, implementer reports, review packages, and the -# progress ledger. Print the directory's absolute path. +# Resolve and ensure the working-tree directory SDD uses for one plan's +# short-lived artifacts: task briefs, implementer reports, review packages, +# and the progress ledger. Print the plan directory's absolute path. +# +# One directory per plan (.superpowers/sdd//) so a follow-up +# plan in the same working tree can never read or overwrite another plan's +# artifacts. A stale ledger misread as current progress makes controllers +# skip whole task sequences — plan-scoping removes that failure structurally. # # The workspace lives in the working tree (not under .git/) because Claude Code # treats .git/ as a protected path and denies agent writes there — which blocks # an implementer subagent from writing its report file. A self-ignoring -# .gitignore keeps the workspace out of `git status` and out of accidental -# commits without modifying any tracked file. +# .gitignore at .superpowers/sdd/ keeps every plan's workspace out of +# `git status` and out of accidental commits without modifying any tracked file. # # Single source of truth for the workspace location, so task-brief and # review-package cannot drift to different directories. # -# Usage: sdd-workspace +# Usage: sdd-workspace PLAN_FILE set -euo pipefail +if [ $# -ne 1 ]; then + echo "usage: sdd-workspace PLAN_FILE" >&2 + exit 2 +fi + +plan=$1 +[ -f "$plan" ] || { echo "no such plan file: $plan" >&2; exit 2; } + +slug=$(basename "$plan" .md) +[ -n "$slug" ] && [ "$slug" != "." ] && [ "$slug" != ".." ] \ + || { echo "cannot derive a workspace name from: $plan" >&2; exit 2; } + root=$(git rev-parse --show-toplevel) -dir="$root/.superpowers/sdd" +base="$root/.superpowers/sdd" +dir="$base/$slug" mkdir -p "$dir" -printf '*\n' > "$dir/.gitignore" +printf '*\n' > "$base/.gitignore" cd "$dir" && pwd diff --git a/.github/skills/superpowers-subagent-driven-development/scripts/task-brief b/.github/skills/superpowers-subagent-driven-development/scripts/task-brief index 247a7670..612e14a1 100755 --- a/.github/skills/superpowers-subagent-driven-development/scripts/task-brief +++ b/.github/skills/superpowers-subagent-driven-development/scripts/task-brief @@ -4,8 +4,9 @@ # through the controller's context. # # Usage: task-brief PLAN_FILE TASK_NUMBER [OUTFILE] -# Default OUTFILE: /.superpowers/sdd/task--brief.md -# (per worktree; concurrent runs in the same working tree share it). +# Default OUTFILE: /.superpowers/sdd//task--brief.md +# (per plan and per worktree; concurrent runs of the SAME plan in the same +# working tree share it). set -euo pipefail if [ $# -lt 2 ] || [ $# -gt 3 ]; then @@ -20,7 +21,7 @@ n=$2 if [ $# -eq 3 ]; then out=$3 else - dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace") + dir=$("$(cd "$(dirname "$0")" && pwd)/sdd-workspace" "$plan") out="$dir/task-${n}-brief.md" fi diff --git a/.github/skills/superpowers-subagent-driven-development/task-reviewer-prompt.md b/.github/skills/superpowers-subagent-driven-development/task-reviewer-prompt.md index 588a4022..fefaea8a 100644 --- a/.github/skills/superpowers-subagent-driven-development/task-reviewer-prompt.md +++ b/.github/skills/superpowers-subagent-driven-development/task-reviewer-prompt.md @@ -178,11 +178,8 @@ Subagent (general-purpose): - `[BASE_SHA]` — commit before this task - `[HEAD_SHA]` — current commit - `[DIFF_FILE]` — REQUIRED: the path the controller wrote the review - package to (`scripts/review-package BASE HEAD` prints the unique path it - wrote; the package never enters the controller's context) + package to (`scripts/review-package PLAN_FILE BASE HEAD` prints the unique + path it wrote; the package never enters the controller's context) **Reviewer returns:** Spec Compliance verdict (✅/❌/⚠️), Strengths, Issues (Critical/Important/Minor), Task quality verdict - -A fix dispatch can address spec gaps and quality findings together; -re-review after fixes covers both verdicts. diff --git a/.github/skills/superpowers-systematic-debugging/SKILL.md b/.github/skills/superpowers-systematic-debugging/SKILL.md index b6d866bb..3b3fba77 100644 --- a/.github/skills/superpowers-systematic-debugging/SKILL.md +++ b/.github/skills/superpowers-systematic-debugging/SKILL.md @@ -7,8 +7,6 @@ description: Use when encountering any bug, test failure, or unexpected behavior ## Overview -Random fixes waste time and create new bugs. Quick patches mask underlying issues. - **Core principle:** ALWAYS find root cause before attempting fixes. Symptom fixes are failure. **Violating the letter of this process is violating the spirit of debugging.** @@ -188,6 +186,7 @@ You MUST complete each phase before proceeding to the next. - Test passes now? - No other tests broken? - Issue actually resolved? + - Use the `superpowers-verification-before-completion` skill before claiming success 4. **If Fix Doesn't Work** - STOP @@ -282,15 +281,3 @@ These techniques are part of systematic debugging and available in this director - **`root-cause-tracing.md`** - Trace bugs backward through call stack to find original trigger - **`defense-in-depth.md`** - Add validation at multiple layers after finding root cause - **`condition-based-waiting.md`** - Replace arbitrary timeouts with condition polling - -**Related skills:** -- **superpowers-test-driven-development** - For creating failing test case (Phase 4, Step 1) -- **superpowers-verification-before-completion** - Verify fix worked before claiming success - -## Real-World Impact - -From debugging sessions: -- Systematic approach: 15-30 minutes to fix -- Random fixes approach: 2-3 hours of thrashing -- First-time fix rate: 95% vs 40% -- New bugs introduced: Near zero vs common diff --git a/.github/skills/superpowers-systematic-debugging/find-polluter.sh b/.github/skills/superpowers-systematic-debugging/find-polluter.sh index 1d71c560..985f5d08 100755 --- a/.github/skills/superpowers-systematic-debugging/find-polluter.sh +++ b/.github/skills/superpowers-systematic-debugging/find-polluter.sh @@ -18,9 +18,18 @@ echo "🔍 Searching for test that creates: $POLLUTION_CHECK" echo "Test pattern: $TEST_PATTERN" echo "" -# Get list of test files -TEST_FILES=$(find . -path "$TEST_PATTERN" | sort) -TOTAL=$(echo "$TEST_FILES" | wc -l | tr -d ' ') +# Get list of test files (find . emits ./-prefixed paths, so accept the +# pattern written with or without a leading ./) +TEST_PATTERN="${TEST_PATTERN#./}" +# find -path can't match '**/' against zero directory levels, so a pattern +# like src/**/*.test.ts would skip src/top.test.ts; also try the pattern +# with '**/' collapsed to cover files directly under the base directory. +TEST_FILES=$(find . \( -path "./$TEST_PATTERN" -o -path "./${TEST_PATTERN//\*\*\//}" \) | sort -u) +if [ -z "$TEST_FILES" ]; then + TOTAL=0 +else + TOTAL=$(printf '%s\n' "$TEST_FILES" | wc -l | tr -d ' ') +fi echo "Found $TOTAL test files" echo "" diff --git a/.github/skills/superpowers-test-driven-development/SKILL.md b/.github/skills/superpowers-test-driven-development/SKILL.md index d3207f67..a4371a00 100644 --- a/.github/skills/superpowers-test-driven-development/SKILL.md +++ b/.github/skills/superpowers-test-driven-development/SKILL.md @@ -203,69 +203,25 @@ Next failing test for next feature. | **Clear** | Name describes behavior | `test('test1')` | | **Shows intent** | Demonstrates desired API | Obscures what code should do | -## Why Order Matters - -**"I'll write tests after to verify it works"** - -Tests written after code pass immediately. Passing immediately proves nothing: -- Might test wrong thing -- Might test implementation, not behavior -- Might miss edge cases you forgot -- You never saw it catch the bug - -Test-first forces you to see the test fail, proving it actually tests something. - -**"I already manually tested all the edge cases"** - -Manual testing is ad-hoc. You think you tested everything but: -- No record of what you tested -- Can't re-run when code changes -- Easy to forget cases under pressure -- "It worked when I tried it" ≠ comprehensive - -Automated tests are systematic. They run the same way every time. - -**"Deleting X hours of work is wasteful"** - -Sunk cost fallacy. The time is already gone. Your choice now: -- Delete and rewrite with TDD (X more hours, high confidence) -- Keep it and add tests after (30 min, low confidence, likely bugs) - -The "waste" is keeping code you can't trust. Working code without real tests is technical debt. - -**"TDD is dogmatic, being pragmatic means adapting"** - -TDD IS pragmatic: -- Finds bugs before commit (faster than debugging after) -- Prevents regressions (tests catch breaks immediately) -- Documents behavior (tests show how to use code) -- Enables refactoring (change freely, tests catch breaks) - -"Pragmatic" shortcuts = debugging in production = slower. - -**"Tests after achieve the same goals - it's spirit not ritual"** - -No. Tests-after answer "What does this do?" Tests-first answer "What should this do?" - -Tests-after are biased by your implementation. You test what you built, not what's required. You verify remembered edge cases, not discovered ones. - -Tests-first force edge case discovery before implementing. Tests-after verify you remembered everything (you didn't). - -30 minutes of tests after ≠ TDD. You get coverage, lose proof tests work. +When writing or changing any test, read [writing-good-tests.md](writing-good-tests.md) for the rules that keep tests honest: +- Name the production change that would make the test fail — before writing it +- Assert on real behavior, never on mock behavior +- Keep test-only code in test utilities, out of production classes +- Understand a dependency's side effects before mocking it ## Common Rationalizations | Excuse | Reality | |--------|---------| | "Too simple to test" | Simple code breaks. Test takes 30 seconds. | -| "I'll test after" | Tests passing immediately prove nothing. | -| "Tests after achieve same goals" | Tests-after = "what does this do?" Tests-first = "what should this do?" | -| "Already manually tested" | Ad-hoc ≠ systematic. No record, can't re-run. | -| "Deleting X hours is wasteful" | Sunk cost fallacy. Keeping unverified code is technical debt. | +| "I'll test after" | Tests written after pass immediately — which proves nothing. They may test the wrong thing, test the implementation instead of the behavior, or miss the edge case you forgot. You never watched it fail, so you never proved it can catch the bug. Test-first forces that failure. | +| "Tests after achieve same goals (spirit not ritual)" | Tests-after answer "what does this do?"; tests-first answer "what should this do?" Tests written after are biased by the code you already wrote — you verify the cases you remembered, not the ones you'd have discovered. Coverage without proof the tests work. | +| "Already manually tested" | Manual testing is ad-hoc: no record of what you covered, no way to re-run it when the code changes, easy to forget cases under pressure. "Worked when I tried it" ≠ comprehensive. Automated tests run the same way every time. | +| "Deleting X hours is wasteful" | Sunk cost fallacy — that time is already spent either way. The real choice: rewrite with TDD (high confidence) vs. keep it and bolt tests on after (low confidence, likely bugs). Keeping code you can't trust is the waste. | | "Keep as reference, write tests first" | You'll adapt it. That's testing after. Delete means delete. | | "Need to explore first" | Fine. Throw away exploration, start with TDD. | | "Test hard = design unclear" | Listen to test. Hard to test = hard to use. | -| "TDD will slow me down" | TDD faster than debugging. Pragmatic = test-first. | +| "TDD will slow me down" | TDD IS the pragmatic path: catches bugs before commit, prevents regressions, lets you refactor without fear. "Pragmatic" shortcuts mean debugging in production — slower, not faster. | | "Manual test faster" | Manual doesn't prove edge cases. You'll re-test every change. | | "Existing code has no tests" | You're improving it. Add tests for existing code. | @@ -354,13 +310,6 @@ Bug found? Write failing test reproducing it. Follow TDD cycle. Test proves fix Never fix bugs without a test. -## Testing Anti-Patterns - -When adding mocks or test utilities, read [testing-anti-patterns.md](testing-anti-patterns.md) to avoid common pitfalls: -- Testing mock behavior instead of real behavior -- Adding test-only methods to production classes -- Mocking without understanding dependencies - ## Final Rule ``` diff --git a/.github/skills/superpowers-test-driven-development/testing-anti-patterns.md b/.github/skills/superpowers-test-driven-development/testing-anti-patterns.md deleted file mode 100644 index e77ab6b6..00000000 --- a/.github/skills/superpowers-test-driven-development/testing-anti-patterns.md +++ /dev/null @@ -1,299 +0,0 @@ -# Testing Anti-Patterns - -**Load this reference when:** writing or changing tests, adding mocks, or tempted to add test-only methods to production code. - -## Overview - -Tests must verify real behavior, not mock behavior. Mocks are a means to isolate, not the thing being tested. - -**Core principle:** Test what the code does, not what the mocks do. - -**Following strict TDD prevents these anti-patterns.** - -## The Iron Laws - -``` -1. NEVER test mock behavior -2. NEVER add test-only methods to production classes -3. NEVER mock without understanding dependencies -``` - -## Anti-Pattern 1: Testing Mock Behavior - -**The violation:** -```typescript -// ❌ BAD: Testing that the mock exists -test('renders sidebar', () => { - render(); - expect(screen.getByTestId('sidebar-mock')).toBeInTheDocument(); -}); -``` - -**Why this is wrong:** -- You're verifying the mock works, not that the component works -- Test passes when mock is present, fails when it's not -- Tells you nothing about real behavior - -**your human partner's correction:** "Are we testing the behavior of a mock?" - -**The fix:** -```typescript -// ✅ GOOD: Test real component or don't mock it -test('renders sidebar', () => { - render(); // Don't mock sidebar - expect(screen.getByRole('navigation')).toBeInTheDocument(); -}); - -// OR if sidebar must be mocked for isolation: -// Don't assert on the mock - test Page's behavior with sidebar present -``` - -### Gate Function - -``` -BEFORE asserting on any mock element: - Ask: "Am I testing real component behavior or just mock existence?" - - IF testing mock existence: - STOP - Delete the assertion or unmock the component - - Test real behavior instead -``` - -## Anti-Pattern 2: Test-Only Methods in Production - -**The violation:** -```typescript -// ❌ BAD: destroy() only used in tests -class Session { - async destroy() { // Looks like production API! - await this._workspaceManager?.destroyWorkspace(this.id); - // ... cleanup - } -} - -// In tests -afterEach(() => session.destroy()); -``` - -**Why this is wrong:** -- Production class polluted with test-only code -- Dangerous if accidentally called in production -- Violates YAGNI and separation of concerns -- Confuses object lifecycle with entity lifecycle - -**The fix:** -```typescript -// ✅ GOOD: Test utilities handle test cleanup -// Session has no destroy() - it's stateless in production - -// In test-utils/ -export async function cleanupSession(session: Session) { - const workspace = session.getWorkspaceInfo(); - if (workspace) { - await workspaceManager.destroyWorkspace(workspace.id); - } -} - -// In tests -afterEach(() => cleanupSession(session)); -``` - -### Gate Function - -``` -BEFORE adding any method to production class: - Ask: "Is this only used by tests?" - - IF yes: - STOP - Don't add it - Put it in test utilities instead - - Ask: "Does this class own this resource's lifecycle?" - - IF no: - STOP - Wrong class for this method -``` - -## Anti-Pattern 3: Mocking Without Understanding - -**The violation:** -```typescript -// ❌ BAD: Mock breaks test logic -test('detects duplicate server', () => { - // Mock prevents config write that test depends on! - vi.mock('ToolCatalog', () => ({ - discoverAndCacheTools: vi.fn().mockResolvedValue(undefined) - })); - - await addServer(config); - await addServer(config); // Should throw - but won't! -}); -``` - -**Why this is wrong:** -- Mocked method had side effect test depended on (writing config) -- Over-mocking to "be safe" breaks actual behavior -- Test passes for wrong reason or fails mysteriously - -**The fix:** -```typescript -// ✅ GOOD: Mock at correct level -test('detects duplicate server', () => { - // Mock the slow part, preserve behavior test needs - vi.mock('MCPServerManager'); // Just mock slow server startup - - await addServer(config); // Config written - await addServer(config); // Duplicate detected ✓ -}); -``` - -### Gate Function - -``` -BEFORE mocking any method: - STOP - Don't mock yet - - 1. Ask: "What side effects does the real method have?" - 2. Ask: "Does this test depend on any of those side effects?" - 3. Ask: "Do I fully understand what this test needs?" - - IF depends on side effects: - Mock at lower level (the actual slow/external operation) - OR use test doubles that preserve necessary behavior - NOT the high-level method the test depends on - - IF unsure what test depends on: - Run test with real implementation FIRST - Observe what actually needs to happen - THEN add minimal mocking at the right level - - Red flags: - - "I'll mock this to be safe" - - "This might be slow, better mock it" - - Mocking without understanding the dependency chain -``` - -## Anti-Pattern 4: Incomplete Mocks - -**The violation:** -```typescript -// ❌ BAD: Partial mock - only fields you think you need -const mockResponse = { - status: 'success', - data: { userId: '123', name: 'Alice' } - // Missing: metadata that downstream code uses -}; - -// Later: breaks when code accesses response.metadata.requestId -``` - -**Why this is wrong:** -- **Partial mocks hide structural assumptions** - You only mocked fields you know about -- **Downstream code may depend on fields you didn't include** - Silent failures -- **Tests pass but integration fails** - Mock incomplete, real API complete -- **False confidence** - Test proves nothing about real behavior - -**The Iron Rule:** Mock the COMPLETE data structure as it exists in reality, not just fields your immediate test uses. - -**The fix:** -```typescript -// ✅ GOOD: Mirror real API completeness -const mockResponse = { - status: 'success', - data: { userId: '123', name: 'Alice' }, - metadata: { requestId: 'req-789', timestamp: 1234567890 } - // All fields real API returns -}; -``` - -### Gate Function - -``` -BEFORE creating mock responses: - Check: "What fields does the real API response contain?" - - Actions: - 1. Examine actual API response from docs/examples - 2. Include ALL fields system might consume downstream - 3. Verify mock matches real response schema completely - - Critical: - If you're creating a mock, you must understand the ENTIRE structure - Partial mocks fail silently when code depends on omitted fields - - If uncertain: Include all documented fields -``` - -## Anti-Pattern 5: Integration Tests as Afterthought - -**The violation:** -``` -✅ Implementation complete -❌ No tests written -"Ready for testing" -``` - -**Why this is wrong:** -- Testing is part of implementation, not optional follow-up -- TDD would have caught this -- Can't claim complete without tests - -**The fix:** -``` -TDD cycle: -1. Write failing test -2. Implement to pass -3. Refactor -4. THEN claim complete -``` - -## When Mocks Become Too Complex - -**Warning signs:** -- Mock setup longer than test logic -- Mocking everything to make test pass -- Mocks missing methods real components have -- Test breaks when mock changes - -**your human partner's question:** "Do we need to be using a mock here?" - -**Consider:** Integration tests with real components often simpler than complex mocks - -## TDD Prevents These Anti-Patterns - -**Why TDD helps:** -1. **Write test first** → Forces you to think about what you're actually testing -2. **Watch it fail** → Confirms test tests real behavior, not mocks -3. **Minimal implementation** → No test-only methods creep in -4. **Real dependencies** → You see what the test actually needs before mocking - -**If you're testing mock behavior, you violated TDD** - you added mocks without watching test fail against real code first. - -## Quick Reference - -| Anti-Pattern | Fix | -|--------------|-----| -| Assert on mock elements | Test real component or unmock it | -| Test-only methods in production | Move to test utilities | -| Mock without understanding | Understand dependencies first, mock minimally | -| Incomplete mocks | Mirror real API completely | -| Tests as afterthought | TDD - tests first | -| Over-complex mocks | Consider integration tests | - -## Red Flags - -- Assertion checks for `*-mock` test IDs -- Methods only called in test files -- Mock setup is >50% of test -- Test fails when you remove mock -- Can't explain why mock is needed -- Mocking "just to be safe" - -## The Bottom Line - -**Mocks are tools to isolate, not things to test.** - -If TDD reveals you're testing mock behavior, you've gone wrong. - -Fix: Test real behavior or question why you're mocking at all. diff --git a/.github/skills/superpowers-test-driven-development/writing-good-tests.md b/.github/skills/superpowers-test-driven-development/writing-good-tests.md new file mode 100644 index 00000000..5691ee03 --- /dev/null +++ b/.github/skills/superpowers-test-driven-development/writing-good-tests.md @@ -0,0 +1,198 @@ +# Writing Good Tests + +**Load this reference when:** writing or changing tests, adding mocks, or +adding cleanup/helper methods for tests. + +## Overview + +A test exists to catch a specific break. Two principles govern everything +here: + +``` +1. Every test names the break it catches +2. Every test exercises the real thing +``` + +Strict TDD produces both naturally: a test written first and watched +failing against real code has already proven it can fail, and only earns +a mock when the real dependency proves slow or external. + +## Principle 1: Name the Break + +Before writing the test body, answer: **what production change should +make this test fail — and is that change a bug or a decision?** A test +earns its place by catching a wrong branch, missing side effect, wrong +argument, boundary case, or broken contract. + +**Derive expectations independently.** Use literals and hand-checked +fixtures; table-driven tests with literal `want` values are the preferred +shape. An expectation computed by the code under test — or its helpers — +passes no matter what that code does: + +```typescript +// ❌ Mirror assertion: the same builder computes both sides — always true +const expected = buildSearchQuery({ tag: 'urgent' }); +expect(buildSearchQuery({ tag: 'urgent' })).toBe(expected); + +// ✅ Hand-derived literal +expect(buildSearchQuery({ tag: 'urgent' })).toBe('tag:"urgent"'); +``` + +**No change detectors.** If only intentional decisions can fail a test — +a constant's value, exact message wording, private structure — it fires +on redesign and sleeps through bugs. Test the behavior that depends on +the decision: not `expect(MAX_RETRIES).toBe(5)` but "a failing call is +retried 5 times and the 6th attempt never happens." + +**Behavior, not text.** Asserting that a script, skill, or config +contains an exact line proves only that the source is the source. Run +scripts against controlled inputs and assert outputs, side effects, or +exit codes. Documents that instruct agents are tested by the consuming +agent's behavior (superpowers-writing-skills); prose for humans earns no +test at all. + +**Your code, not the framework.** Test the contract your code makes at +its boundaries — the route you register, the query you emit, the payload +you produce. Upstream mechanics are their maintainers' tests to write +(the classic: asserting your router invokes a registered handler — that +is the framework's test, not yours). When upstream behavior genuinely +surprised you, write one narrow characterization test naming the +assumption. The same boundary applies inside your code: constructors, +getters, constants, and trivial forwarding earn tests only when they +validate, normalize, default, derive, enforce, or cause side effects — +otherwise assert the first consumer-visible result that depends on them. + +### Gate Function + +``` +BEFORE writing the test body: + Name the production change that would make this test fail. + + Cannot name one → redesign around an observable behavior + "The source text changed" → run the artifact and assert its effects + Only intentional decisions → change detector; test the behavior + that depends on the decision + + Confirm the expected value is derived without the code under test. + IF it reuses the code's logic or helpers: + Replace it with a literal or hand-checked fixture +``` + +## Principle 2: Exercise the Real Thing + +**The mock earns no assertions.** A mock assertion passes when the mock +is present and fails when it is absent — it says nothing about the +component. Assert the real component's behavior; if the mock is what you +are checking, unmock it or delete the assertion. + +```typescript +// ✅ Real behavior +expect(screen.getByRole('navigation')).toBeInTheDocument(); + +// ❌ Mock existence +expect(screen.getByTestId('sidebar-mock')).toBeInTheDocument(); +``` + +**your human partner's correction:** "Are we testing the behavior of a +mock?" + +**Mock at the right level.** Learn every side effect of the real method +before replacing it; mock the slow or external operation and keep what +the test depends on real. When unsure, run the test against the real +implementation first and observe what actually needs to happen. + +```typescript +// ❌ The mock swallows the config write that duplicate detection reads +vi.mock('ToolCatalog', () => ({ + discoverAndCacheTools: vi.fn().mockResolvedValue(undefined) +})); + +// ✅ Mock only the slow server startup; the config write stays real +vi.mock('MCPServerManager'); +``` + +**Make doubles specific.** When arguments, call counts, or ordering are +part of the contract, assert them — a fake that accepts anything verifies +nothing. Give each branch (success, error, malformed) its own fixture or +spy, so the wrong branch cannot satisfy the expectation. + +**Mirror real data completely.** Mock the complete structure as it exists +in reality — all documented fields — not just the ones your test reads. +Partial mocks fail silently when downstream code reads an omitted field: +the test passes while integration breaks. + +**Production classes carry production methods only.** Cleanup that only +tests need lives in test utilities, never as a `destroy()` on the +production class. Ask: is this method called only from tests? Does this +class own this resource's lifecycle? Wrong answers → test utility. + +**Prefer real components over complex mocks.** When mock setup outgrows +the test logic, mocks miss methods the real components have, or tests +break when the mock changes, switch to an integration test with real +components. **your human partner's question:** "Do we need to be using a +mock here?" + +### Gate Function + +``` +BEFORE adding a mock or test helper: + List the real method's side effects; keep the ones the test + depends on real — mock the slow/external level below them. + + Mock responses mirror the complete real structure. + + A method only tests call lives in test utilities, not production. + + About to assert on the mock itself? + Unmock it or delete the assertion. +``` + +## Tests Ship With the Implementation + +The TDD cycle — failing test, minimal implementation, refactor — is what +"complete" means. Ship the tests the behavior needs and only those: +trivial code and human prose earn none, and a test written to satisfy +process costs maintenance forever. + +## The Mutation Check + +Before finishing, mentally mutate the production code; at least one test +should fail for each realistic mutation: + +- Wrong constant or argument +- Wrong branch handler +- Missing state change or side effect +- Empty or default return +- Missing validation for zero, empty, nil, unauthorized, or malformed input + +A mutation nothing catches marks the behavior as unprotected — or the +test as tautological. + +## Quick Reference + +| When you... | Do | +|-------------|-----| +| Write any test | Name the break it catches — a bug, not a decision | +| Build an expected value | Derive it by hand; never with the code under test | +| Test a script or document | Run it / pressure-test its consumer; never grep its text | +| Reach for a dependency test | Test your boundary contract, not their documented mechanics | +| Want to assert on a mocked element | Test the real component, or unmock it | +| Are about to mock a method | Learn its side effects; mock the slow/external level | +| Build a mock response | Mirror the real structure completely | +| Need cleanup only tests use | Put it in test utilities | +| Watch mock setup balloon | Switch to an integration test with real components | +| Finish a test file | Run the mutation check | + +## Warning Signs + +- Setup and assertion share the same object, guaranteeing equality +- The test can fail only through a panic, crash, or missing selector +- The test fails on every intentional change, never on accidental breakage +- Expected values are hidden behind loops, builders, or helpers +- The test greps source text, or asserts a removed symbol stays removed +- The test would still matter if only the framework remained +- The test exists for coverage, checking no side effect or outcome +- An assertion checks a `*-mock` test ID, or fails if you remove the mock +- A method is called only from test files +- Mock setup is more than half the test, or you can't explain why the mock is needed +- Mocking "just to be safe" diff --git a/.github/skills/superpowers-using-git-worktrees/SKILL.md b/.github/skills/superpowers-using-git-worktrees/SKILL.md index e738ece1..462b8750 100644 --- a/.github/skills/superpowers-using-git-worktrees/SKILL.md +++ b/.github/skills/superpowers-using-git-worktrees/SKILL.md @@ -156,47 +156,12 @@ Ready to implement | Tests fail during baseline | Report failures + ask | | No package.json/Cargo.toml | Skip dependency install | -## Common Mistakes - -### Fighting the harness - -- **Problem:** Using `git worktree add` when the platform already provides isolation -- **Fix:** Step 0 detects existing isolation. Step 1a defers to native tools. - -### Skipping detection - -- **Problem:** Creating a nested worktree inside an existing one -- **Fix:** Always run Step 0 before creating anything - -### Skipping ignore verification - -- **Problem:** Worktree contents get tracked, pollute git status -- **Fix:** Always use `git check-ignore` before creating project-local worktree - -### Assuming directory location - -- **Problem:** Creates inconsistency, violates project conventions -- **Fix:** Follow priority: explicit instructions > existing project-local directory > default - -### Proceeding with failing tests - -- **Problem:** Can't distinguish new bugs from pre-existing issues -- **Fix:** Report failures, get explicit permission to proceed - -## Red Flags - -**Never:** -- Create a worktree when Step 0 detects existing isolation -- Use `git worktree add` when you have a native worktree tool (e.g., `EnterWorktree`). This is the #1 mistake — if you have it, use it. -- Skip Step 1a by jumping straight to Step 1b's git commands -- Create worktree without verifying it's ignored (project-local) -- Skip baseline test verification -- Proceed with failing tests without asking - -**Always:** -- Run Step 0 detection first -- Prefer native tools over git fallback -- Follow directory priority: explicit instructions > existing project-local directory > default -- Verify directory is ignored for project-local -- Auto-detect and run project setup -- Verify clean test baseline +## Common Rationalizations + +| Excuse | Reality | +|--------|---------| +| "I'm obviously not in a worktree — no need to check" | Run Step 0. Harness-created isolation and submodules both fool eyeballing; the detection commands settle it. | +| "`git worktree add` is quicker than hunting for a native tool" | A native tool (e.g. `EnterWorktree`) owns placement, branching, and cleanup. Bypassing it is the #1 mistake — it creates phantom state your harness can't see or manage. | +| "The worktree directory is surely ignored already" | Run `git check-ignore`. An unignored worktree directory commits the whole tree into the repo. | +| "Any directory name works" | Explicit instructions beat an existing project-local directory, which beats the `.worktrees/` default. | +| "The workspace is fresh — baseline tests can wait" | A dirty baseline makes every later failure ambiguous. Run the tests now; proceeding past failures is your human partner's call. | diff --git a/.github/skills/superpowers-using-superpowers/references/antigravity-tools.md b/.github/skills/superpowers-using-superpowers/references/antigravity-tools.md index 71155fde..2e1eac90 100644 --- a/.github/skills/superpowers-using-superpowers/references/antigravity-tools.md +++ b/.github/skills/superpowers-using-superpowers/references/antigravity-tools.md @@ -4,7 +4,7 @@ Skills speak in actions ("dispatch a subagent", "create a todo", "read a file"). | Action skills request | Antigravity CLI equivalent | |----------------------|----------------------| -| Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_subagent` with a built-in `TypeName` — `self` for full-capability work, `research` for read-only (see [Subagent support](#subagent-support)) | +| Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_subagent` with a built-in `TypeName` — `self` for full-capability work, `research` for read-only | | Task tracking ("create a todo", "mark complete") | a **task artifact** — `write_to_file` with `IsArtifact: true` and `ArtifactType: "task"` (see [Task tracking](#task-tracking)). **Not** `manage_task`, which manages background processes. | ## Task tracking diff --git a/.github/skills/superpowers-using-superpowers/references/codex-tools.md b/.github/skills/superpowers-using-superpowers/references/codex-tools.md index 1897cc3b..b14b5858 100644 --- a/.github/skills/superpowers-using-superpowers/references/codex-tools.md +++ b/.github/skills/superpowers-using-superpowers/references/codex-tools.md @@ -7,7 +7,7 @@ Add to your Codex config (`~/.codex/config.toml`): multi_agent = true ``` -This enables `spawn_agent`, `wait_agent`, and `close_agent` for skills like `dispatching-parallel-agents` and `subagent-driven-development`. When using subagent-driven-development, you should always close implementer and reviewer subagents when they have finished all their work. +This enables `spawn_agent`, `wait_agent`, and `close_agent` for skills like `dispatching-parallel-agents` and `subagent-driven-development`. When using subagent-driven-development, close reviewer subagents when their review returns. Keep each implementer subagent open until its task's review passes — the fix loop resumes the implementer — then close it. If your harness cannot send another message to a spawned agent, dispatch each fix round as a fresh implementer carrying the brief, the report file, and the findings. ## Environment Detection diff --git a/.github/skills/superpowers-using-superpowers/references/gemini-tools.md b/.github/skills/superpowers-using-superpowers/references/gemini-tools.md new file mode 100644 index 00000000..8f1a79cf --- /dev/null +++ b/.github/skills/superpowers-using-superpowers/references/gemini-tools.md @@ -0,0 +1,63 @@ +# Gemini CLI Tool Mapping + +Skills speak in actions ("dispatch a subagent", "create a todo", "read a file"). On Gemini CLI these resolve to the tools below. + +| Action skills request | Gemini CLI equivalent | +|----------------------|----------------------| +| Read a file | `read_file` | +| Read multiple files at once | `read_many_files` | +| Create a new file | `write_file` | +| Edit a file | `replace` | +| Run a shell command | `run_shell_command` | +| Search file contents | `grep_search` | +| Find files by name | `glob` | +| List files and subdirectories | `list_directory` | +| Fetch a URL | `web_fetch` | +| Search the web | `google_web_search` | +| Invoke a skill | `activate_skill` | +| Dispatch a subagent (`Subagent (general-purpose):` template) | `invoke_agent` with `agent_name: "generalist"` (invocable via `@generalist` chat syntax — see [Subagent support](#subagent-support)) | +| Multiple parallel dispatches | Multiple `invoke_agent` calls in the same response | +| Task tracking ("create a todo", "mark complete") | `write_todos` (statuses: pending, in_progress, completed, cancelled, blocked) | + +## Instructions file + +When a skill mentions "your instructions file", on Gemini CLI this is **`GEMINI.md`**. Gemini CLI loads `GEMINI.md` hierarchically: global at `~/.gemini/GEMINI.md`, project-level files in workspace directories and their ancestors, and sub-directory `GEMINI.md` files when a tool accesses files in those directories. + +## Personal skills directory + +User-level skills live at **`~/.gemini/skills/`**, with **`~/.agents/skills/`** as a cross-runtime alias (shared with Codex and Copilot CLI). When both directories exist at the same scope, `.agents/skills/` takes precedence. Each skill is a subdirectory containing a `SKILL.md` (with `name` and `description` frontmatter). + +## Subagent support + +Gemini CLI dispatches subagents through the `invoke_agent` tool, which takes `agent_name` and `prompt` parameters. The same dispatch is also surfaced as a chat-syntax shortcut: typing `@generalist ` is equivalent to calling `invoke_agent` with `agent_name: "generalist"`. Built-in agent names include `generalist`, `cli_help`, `codebase_investigator`, and (with browser tooling enabled) `browser_agent`. + +Skills dispatch with `Subagent (general-purpose):` and either reference a prompt-template file (e.g., `superpowers-subagent-driven-development`'s `./implementer-prompt.md`) or supply an inline prompt. On Gemini CLI: + +| Skill dispatch form | Gemini CLI equivalent | +|---------------------|----------------------| +| References a `*-prompt.md` template (implementer, task-reviewer, code-reviewer, etc.) | Fill the template, then `invoke_agent` with `agent_name: "generalist"` and the filled prompt | +| References `superpowers-requesting-code-review`'s `./code-reviewer.md` | `invoke_agent` with `agent_name: "generalist"` and the filled review template | +| Inline prompt (no template referenced) | `invoke_agent` with `agent_name: "generalist"` and your inline prompt | + +### Prompt filling + +Skills provide prompt templates with placeholders like `{WHAT_WAS_IMPLEMENTED}` or `[FULL TEXT of task]`. Fill all placeholders before passing the complete prompt to `invoke_agent`. The prompt template itself contains the agent's role, review criteria, and expected output format — the subagent will follow it. + +### Parallel dispatch + +Gemini CLI supports parallel subagent dispatch. Issue multiple `invoke_agent` calls in the same response (or multiple `@generalist` invocations in one prompt) to run independent subagent work in parallel. Keep dependent tasks sequential, but do not serialize independent subagent tasks just to preserve a simpler history. + +## Additional Gemini CLI tools + +These tools are unique to Gemini CLI: + +| Tool | Purpose | +|------|---------| +| `save_memory` (legacy) | Persist facts across sessions when `experimental.memoryV2 = false` | +| `get_internal_docs` | Look up Gemini CLI's bundled documentation | +| `ask_user` | Pose structured questions to the user (text / single-select / multi-select) | +| `enter_plan_mode` / `exit_plan_mode` | Switch into and out of read-only plan mode | +| `update_topic` | Update the current conversation's topic / strategic-intent metadata | +| `complete_task` | Signal that a Gemini subagent has completed and return its result to the parent agent | +| `tracker_create_task`, `tracker_update_task`, `tracker_get_task`, `tracker_list_tasks`, `tracker_add_dependency`, `tracker_visualize` | Rich task tracker with dependency and visualization support | +| `read_mcp_resource`, `list_mcp_resources` | MCP resource access | diff --git a/.github/skills/superpowers-verification-before-completion/SKILL.md b/.github/skills/superpowers-verification-before-completion/SKILL.md index b60f91de..4a037c05 100644 --- a/.github/skills/superpowers-verification-before-completion/SKILL.md +++ b/.github/skills/superpowers-verification-before-completion/SKILL.md @@ -7,8 +7,6 @@ description: Use when about to claim work is complete, fixed, or passing, before ## Overview -Claiming work is complete without verification is dishonesty, not efficiency. - **Core principle:** Evidence before claims, always. **Violating the letter of this rule is violating the spirit of this rule.** @@ -105,15 +103,6 @@ Skip any step = lying, not verifying ❌ Trust agent report ``` -## Why This Matters - -From 24 failure memories: -- your human partner said "I don't believe you" - trust broken -- Undefined functions shipped - would crash -- Missing requirements shipped - incomplete features -- Time wasted on false completion → redirect → rework -- Violates: "Honesty is a core value. If you lie, you'll be replaced." - ## When To Apply **ALWAYS before:** @@ -129,11 +118,3 @@ From 24 failure memories: - Paraphrases and synonyms - Implications of success - ANY communication suggesting completion/correctness - -## The Bottom Line - -**No shortcuts for verification.** - -Run the command. Read the output. THEN claim the result. - -This is non-negotiable. diff --git a/.github/skills/superpowers-writing-plans/SKILL.md b/.github/skills/superpowers-writing-plans/SKILL.md index 8ffc1163..68808218 100644 --- a/.github/skills/superpowers-writing-plans/SKILL.md +++ b/.github/skills/superpowers-writing-plans/SKILL.md @@ -135,12 +135,6 @@ Every step must contain the actual content an engineer needs. These are **plan f - Steps that describe what to do without showing how (code blocks required for code steps) - References to types, functions, or methods not defined in any task -## Remember -- Exact file paths always -- Complete code in every step — if a step changes code, show the code -- Exact commands with expected output -- DRY, YAGNI, TDD, frequent commits - ## Self-Review After writing the complete plan, look at the spec with fresh eyes and check the plan against it. This is a checklist you run yourself — not a subagent dispatch. diff --git a/.github/skills/terraform-terraform-search-import/SKILL.md b/.github/skills/terraform-terraform-search-import/SKILL.md index f46669ad..52924140 100644 --- a/.github/skills/terraform-terraform-search-import/SKILL.md +++ b/.github/skills/terraform-terraform-search-import/SKILL.md @@ -40,7 +40,7 @@ Discover existing cloud resources using declarative queries and generate configu - ** If supported**: Check for terraform version available. - ** If terraform version is above 1.14.0** Use Terraform Search workflow (below) - ** If not supported or terraform version is below 1.14.0 **: Use Manual Discovery workflow (see [references/MANUAL-IMPORT.md](references/MANUAL-IMPORT.md)) - + **Note**: The list of supported resources is rapidly expanding. Always verify current support before using manual import. ## Prerequisites @@ -147,7 +147,7 @@ list "aws_instance" "all" { # Find instances by tag list "aws_instance" "production" { provider = aws - + config { filter { name = "tag:Environment" @@ -159,7 +159,7 @@ list "aws_instance" "production" { # Find instances by type list "aws_instance" "large" { provider = aws - + config { filter { name = "instance-type" @@ -183,7 +183,7 @@ locals { list "aws_instance" "all_regions" { for_each = toset(local.regions) provider = aws - + config { region = each.value } @@ -200,7 +200,7 @@ variable "target_environment" { list "aws_instance" "by_env" { provider = aws - + config { filter { name = "tag:Environment" @@ -278,7 +278,7 @@ resource "aws_instance" "web_server" { ami = var.ami_id instance_type = var.instance_type subnet_id = var.subnet_id - + tags = { Name = "web-server" Environment = var.environment @@ -345,7 +345,7 @@ provider "aws" { list "aws_instance" "team_instances" { provider = aws - + config { filter { name = "tag:Owner" @@ -356,7 +356,7 @@ list "aws_instance" "team_instances" { values = ["running"] } } - + limit = 50 } ``` diff --git a/.github/skills/vercel-find-skills/SKILL.md b/.github/skills/vercel-find-skills/SKILL.md index 739a5f90..1d2e5134 100644 --- a/.github/skills/vercel-find-skills/SKILL.md +++ b/.github/skills/vercel-find-skills/SKILL.md @@ -26,7 +26,6 @@ The Skills CLI (`npx skills`) is the package manager for the open agent skills e - `npx skills find [query] [--owner ]` - Search for skills interactively or by keyword, optionally scoped to a GitHub owner - `npx skills add ` - Install a skill from GitHub or other sources -- `npx skills check` - Check for skill updates - `npx skills update` - Update all installed skills **Browse skills at:** https://skills.sh/ diff --git a/.github/workflows/_code-analysis.yml b/.github/workflows/_code-analysis.yml index f0df13eb..9aa3a59d 100644 --- a/.github/workflows/_code-analysis.yml +++ b/.github/workflows/_code-analysis.yml @@ -75,7 +75,7 @@ jobs: "${TEST_VENV}/bin/python" ".github/scripts/audit_copilot_catalog.py" --help "${TEST_VENV}/bin/python" ".github/scripts/detect_token_risks.py" --help "${TEST_VENV}/bin/python" ".github/scripts/validate_internal_skills.py" --help - "${TEST_VENV}/bin/python" ".github/scripts/sync_copilot_catalog.py" --help + "${TEST_VENV}/bin/python" ".github/skills/local-sync-repos/scripts/sync_repos.py" --help "${TEST_VENV}/bin/python" ".github/scripts/github_catalog_validation.py" --help - name: 🔥 Smoke test Bash entry points diff --git a/.markdownlint-cli2.jsonc b/.markdownlint-cli2.jsonc index f71135ea..ece0204a 100644 --- a/.markdownlint-cli2.jsonc +++ b/.markdownlint-cli2.jsonc @@ -11,6 +11,8 @@ "tools/analyze_copilot_debug_log/.venv/**", // Test fixtures are intentionally non-document-shaped markdown inputs. "tests/fixtures/critical_output_*.md", + // Skill bundle card fixtures are emoji cards, not structured Markdown documents. + ".github/skills/internal-gateway-critical-master/fixtures/*.md", // Ignore local OpenCode runtime artifacts. ".opencode/**", // Preserve imported support skill families verbatim by default; lint only repo-owned Markdown. diff --git a/AGENTS.local.md b/AGENTS.local.md new file mode 100644 index 00000000..806b8428 --- /dev/null +++ b/AGENTS.local.md @@ -0,0 +1,38 @@ +# AGENTS.local.md - Repository-Local Policy + +- Do not duplicate skill-owned paths, templates, workflow states, or command + examples in this file. +- Keep volatile inventory out of this file; `.github/INVENTORY.md` owns the live + catalog for this repository. + +## Purpose + +- Keep standards-repository-only policy separate from the shared baseline. +- Keep architecture and local context in `docs/` knowledge documents. + +This file applies only to this standards repository. Do not treat these rules +as consumer-repository defaults without an explicit sync contract change. + +## Standards Repository Role + +- This repository owns the shared Copilot customization baseline, governance + contracts, catalog automation, source-side sync tooling, and the source + content used to generate the global home agent baseline. +- Source-managed AI assets live mainly under `.github/`. + +## Standards Repository Validation + +- Run `make token-risks` or + `python3 ./.github/scripts/detect_token_risks.py --root .` after changes that + affect root policy or major AI assets in this repository. + +## Standards Repository Locality + +- Repo-local planning, brainstorming, temporary analysis, and working artifacts + stay outside `docs/` unless a narrower owner explicitly says otherwise. +- Consumer or target repositories own their local override layers after + materialization. +- Skill bundles under `.github/skills/**` must be self-contained. Keep + instructions, references, examples, fixtures, scripts, and `agents/openai.yaml` + resolvable from the bundle itself; do not require bundle users to load + guidance from outside the skill directory. diff --git a/AGENTS.md b/AGENTS.md index 91e59ae5..67048607 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -4,11 +4,6 @@ agents in this repository. Keep it compact: it should route agents to the nearest owner, avoid duplicated guidance, and require explicit validation. -`` - -This block is the portable source baseline used to generate -`~/.agents/AGENTS.md` for cross-repository use. - ## First Move - Identify the requested target and nearest owner before broad reading. @@ -46,6 +41,7 @@ This block is the portable source baseline used to generate - `AGENTS.md` owns stable repository-wide policy, precedence, tactical defaults, ownership boundaries, and routing anchors. +- `.github/INVENTORY.md` is the exact live inventory of the GitHub Copilot catalog. - Do not put long operational procedures, detailed checklists, detailed file-shape recipes, command playbooks, or tool-specific workflows here. - Short, globally safe best-practice defaults may live here when they improve @@ -121,45 +117,7 @@ source files when (a) modifying/debugging specific code, (b) the graph lacks the Type `/graphify` in Copilot Chat to build or update the graph. -`` - -`` - -- Do not duplicate skill-owned paths, templates, workflow states, or command - examples in this file. -- Keep volatile inventory out of this file; `.github/INVENTORY.md` owns the live - catalog for this repository. - -## Purpose - -- Keep standards-repository-only policy separate from the shared baseline. -- Keep architecture and local context in `docs/` knowledge documents. - -This block applies only to this standards repository. Do not treat these rules -as consumer-repository defaults without an explicit sync contract change. - -## Standards Repository Role - -- This repository owns the shared Copilot customization baseline, governance - contracts, catalog automation, source-side sync tooling, and the source - content used to generate the global home agent baseline. -- Source-managed AI assets live mainly under `.github/`. - -## Standards Repository Validation - -- Run `make token-risks` or - `python3 ./.github/scripts/detect_token_risks.py --root .` after changes that - affect root policy or major AI assets in this repository. - -## Standards Repository Locality - -- Repo-local planning, brainstorming, temporary analysis, and working artifacts - stay outside `docs/` unless a narrower owner explicitly says otherwise. -- Consumer or target repositories own their local override layers after - materialization. -- Skill bundles under `.github/skills/**` must be self-contained. Keep - instructions, references, examples, fixtures, scripts, and `agents/openai.yaml` - resolvable from the bundle itself; do not require bundle users to load - guidance from outside the skill directory. +## Optional Repository-Local Policy -`` +If `AGENTS.local.md` exists next to this file, load and apply it after this +baseline. If it does not exist, continue without error. diff --git a/INTERNAL_CONTRACT.md b/INTERNAL_CONTRACT.md index 363e2845..d557d74c 100644 --- a/INTERNAL_CONTRACT.md +++ b/INTERNAL_CONTRACT.md @@ -34,11 +34,13 @@ Treat the current skill-first architecture as the source of truth. Do not infer - Goal: preserve the current root-policy shape without restoring the retired bridge or context-routing model. - Scope: - root `AGENTS.md` + - root `AGENTS.local.md` - `INTERNAL_CONTRACT.md` - source-side contract tests - Expected behavior: - - root `AGENTS.md` may carry a portable `` block that serves as source content for the generated global `~/.agents/AGENTS.md` baseline - - root `AGENTS.md` may carry a source-local `` block that remains non-portable by default + - root `AGENTS.md` may carry the portable shared baseline used as source content for the generated global `~/.agents/AGENTS.md` baseline + - root `AGENTS.md` may reference optional `AGENTS.local.md` policy that is loaded only when present and remains non-portable by default + - root `AGENTS.local.md` owns standards-repository-only policy and is not synchronized to home runtimes - root `AGENTS.md` may keep compact graph orientation rules in the shared baseline when those rules are globally safe and conditionally worded - tests and validators must align to the current on-disk root-policy shape instead of assuming a separate `## Context Routing` section or the absence of root-level graph guidance @@ -271,11 +273,12 @@ Treat the current skill-first architecture as the source of truth. Do not infer - shared operating-model skills - Expected behavior: - `internal-gateway-idea`, `internal-gateway-review`, `internal-gateway-simple-task`, and `internal-gateway-critical-master` remain the canonical repository-owned skill-first gateway core - - `internal-gateway-idea`, `internal-gateway-review`, `internal-gateway-simple-task`, and `internal-gateway-critical-master` remain the current Copilot wrapper entrypoints for that core + - `internal-gateway-idea`, `internal-gateway-review-generic`, `internal-gateway-simple-task`, and `internal-gateway-critical-master` remain the current Copilot wrapper entrypoints for that core - the default operational model uses direct owner selection or user-selected gateway skills with visible phases instead of a hidden repository-owned front-door router - retained execution stays separate: `internal-gateway-simple-task` consumes approved `compact` plans and `internal-gateway-execute-plans` consumes approved `extended` plans - ambiguous or mixed-shape entry fails safe to `internal-gateway-idea` - unclear target state and multiple credible paths are explicit planning triggers + - `internal-gateway-codebase-improvement` is a manual-only specialist gateway outside the canonical gateway core; it must not become an implicit fallback, peer-dispatch target, or replacement for an existing gateway owner - wrapper owners define boundaries and recommendations instead of active delegation - wrapper owners are not subagent-invoked by default, so hidden peer dispatch stays opt-in and explicit - critical challenge can return reformulation, simple, execute, review, continue-critical, or accept-with-risk outcomes diff --git a/LESSONS_LEARNED.md b/LESSONS_LEARNED.md index e3c9a42c..db1f5922 100644 --- a/LESSONS_LEARNED.md +++ b/LESSONS_LEARNED.md @@ -22,10 +22,10 @@ This file retains durable lessons discovered while completing tasks in this repo | Date | Lesson | Status | Codification target | | --- | --- | --- | --- | -| 2026-04-30 | After `apply`, consumer repos can still show synced files as dirty in the working tree; treat the sync as converged only when planner `managed_mutation_paths` is empty and no managed `create`, `update`, `ensure`, `rebuild`, or `delete` actions remain. | Pending | `.github/skills/local-agent-sync-global-copilot-configs-into-repo/references/sync-contract.md` | -| 2026-04-30 | `render_synced_lessons` must append the no-pending marker (and any required leading blank line) to `section_suffix` after the table data, not to `pending_section_lines` before it; otherwise the marker lands on the same line as the separator and the lessons file diverges from the source template. | Pending | `.github/scripts/lib/syncing.py` | +| 2026-04-30 | After `apply`, consumer repos can still show synced files as dirty in the working tree; treat the sync as converged only when planner `managed_mutation_paths` is empty and no managed `create`, `update`, `ensure`, `rebuild`, or `delete` actions remain. | Pending | `.github/skills/local-sync-repos/references/sync-contract.md` | +| 2026-04-30 | `render_synced_lessons` must append the no-pending marker (and any required leading blank line) to `section_suffix` after the table data, not to `pending_section_lines` before it; otherwise the marker lands on the same line as the separator and the lessons file diverges from the source template. | Pending | `.github/scripts/build_inventory.py` | | 2026-04-30 | When opening multi-repo PRs, GitHub Advanced Security `CodeQL` may surface pre-existing alerts on the PR check with a "code changes were too large" notice; resolve by fixing real findings (e.g. redacting secret values from log strings) and dismissing genuine false positives via `gh api -X PATCH .../code-scanning/alerts/{n}` so the PR check turns NEUTRAL/SUCCESS without merging unrelated noise. | Pending | `cloud-strategy.github/docs/operations/multi-repo-pr-runbook.md` | -| 2026-05-01 | When the new `_pre-commit.yml` workflow is rolled out via copilot-sync, repos that still carry the legacy `terraform-pre-commit.yml` end up with two workflows sharing `concurrency.group: pre-commit-${{ github.ref }}` + `cancel-in-progress: true`, so each push cancels the other and the `pre-commit` PR check stays stuck on `cancelled`. Detect this by listing `.github/workflows/*pre-commit*.yml` per repo and delete the legacy file (or scope the concurrency group to `${{ github.workflow }}`). | Pending | `.github/skills/internal-agent-sync-global-copilot-configs-into-repo/references/sync-contract.md` | +| 2026-05-01 | When the new `_pre-commit.yml` workflow is rolled out via copilot-sync, repos that still carry the legacy `terraform-pre-commit.yml` end up with two workflows sharing `concurrency.group: pre-commit-${{ github.ref }}` + `cancel-in-progress: true`, so each push cancels the other and the `pre-commit` PR check stays stuck on `cancelled`. Detect this by listing `.github/workflows/*pre-commit*.yml` per repo and delete the legacy file (or scope the concurrency group to `${{ github.workflow }}`). | Pending | `.github/skills/local-sync-repos/references/sync-contract.md` | | 2026-05-01 | `terraform_validate` in the pre-commit-terraform Docker image fails on Linux runners with "the cached package for ... does not match any of the checksums recorded in the dependency lock file" when `.terraform.lock.hcl` only contains the `h1:` hash from the platform where it was generated (typically `darwin_arm64`). Re-lock for all consumed platforms with `terraform providers lock -platform=linux_amd64 -platform=darwin_amd64 -platform=darwin_arm64` in every directory that gets `terraform init`-ed by the hook. | Pending | `.github/skills/internal-terraform/references/structure-standard.md` | | 2026-05-01 | After copilot-sync edits Python files, run `pre-commit run ruff-format --all-files` (or push and let CI auto-format) before opening the PR: `ruff-format` will collapse multi-line `logger.error(f"...")` calls onto one line and fail the hook with `files were modified by this hook`. The fix is mechanical, but it surfaces as a red `pre-commit` check until applied. | Pending | `.github/skills/internal-python/SKILL.md` | | 2026-05-01 | When triaging PR checks across many repos, prefer `gh pr checks` text output over `--json name,state,conclusion`: the JSON form silently returns `[]` for some workflow runs (cancelled/queued), while the default text output reliably surfaces the per-check state, duration, and run URL needed for triage. | Pending | `cloud-strategy.github/docs/operations/multi-repo-pr-runbook.md` | diff --git a/Makefile b/Makefile index cc7de857..6472991d 100644 --- a/Makefile +++ b/Makefile @@ -68,7 +68,7 @@ skill-lint: scripts-bootstrap @$(SCRIPTS_RUNNER) validate_internal_skills --root . --strict critical-validate: scripts-bootstrap - @$(SCRIPTS_RUNNER) validate_critical_output --file .github/skills/internal-gateway-critical-master/fixtures/critical_output_valid.md + @$(SCRIPTS_RUNNER) validate_critical_output --strict --file .github/skills/internal-gateway-critical-master/fixtures/critical_output_valid.md internal-gateway-idea-fast-check: scripts-bootstrap @$(PYTHON) .github/skills/internal-gateway-idea/scripts/audit_workflow.py @@ -97,10 +97,13 @@ PLAN_AUTHORING_CLI := .github/skills/internal-gateway-writing-plans/scripts/plan PLAN_EXECUTING_CLI := .github/skills/internal-gateway-execute-plans/scripts/plan_execution.py retained-plan-check: python-version-check - @if [ -z "$(PLAN_FOLDER)" ]; then printf '%s\n' 'PLAN_FOLDER is required (e.g. make retained-plan-check PLAN_FOLDER=tmp/superpowers/my-plan)' >&2; exit 1; fi @case "$(PLAN_STAGE)" in \ - handoff) $(PYTHON) $(PLAN_AUTHORING_CLI) handoff-check "$(PLAN_FOLDER)" ;; \ - completion) $(PYTHON) $(PLAN_EXECUTING_CLI) completion-check "$(PLAN_FOLDER)" ;; \ + handoff) \ + if [ -z "$(PLAN_FOLDER)" ]; then printf '%s\n' 'PLAN_FOLDER is required for handoff stage (e.g. make retained-plan-check PLAN_STAGE=handoff PLAN_FOLDER=tmp/superpowers/plans/my-plan)' >&2; exit 1; fi; \ + $(PYTHON) $(PLAN_AUTHORING_CLI) handoff-check "$(PLAN_FOLDER)" ;; \ + completion) \ + if [ -z "$(PLAN_FILE)" ] || [ -z "$(STATUS_FILE)" ]; then printf '%s\n' 'PLAN_FILE and STATUS_FILE are required for completion stage (e.g. make retained-plan-check PLAN_STAGE=completion PLAN_FILE=tmp/superpowers/plans/my-plan.md STATUS_FILE=tmp/superpowers/plans/my-plan.DONE.md)' >&2; exit 1; fi; \ + $(PYTHON) $(PLAN_EXECUTING_CLI) completion-check "$(PLAN_FILE)" "$(STATUS_FILE)" --repo-root . ;; \ *) printf '%s\n' 'PLAN_STAGE must be handoff or completion.' >&2; exit 1 ;; \ esac diff --git a/docs/tech.md b/docs/tech.md index 96ef28eb..b508b7fa 100644 --- a/docs/tech.md +++ b/docs/tech.md @@ -18,7 +18,7 @@ repository. | Category | Observed tooling | | --- | --- | -| Sync logic | `.github/scripts/sync_copilot_catalog.py`, `.github/scripts/lib/syncing.py`, `./.github/scripts/run.sh ` | +| Sync logic | `.github/skills/local-sync-repos/scripts/sync_repos.py`, `.github/skills/local-sync-repos/scripts/sync_contract.py` | | Catalog generation | `.github/scripts/build_inventory.py`, inventory helpers in `.github/scripts/lib/`, and `./.github/scripts/run.sh build_inventory` | | Contract checks | `pytest` tests under `tests/`, plus validator subcommands through `run.sh` | | Linting and policy checks | Make targets such as `docs-lint`, `token-risks`, and catalog checks | diff --git a/tests/github/scripts/lib/test_inventory.py b/tests/github/scripts/lib/test_inventory.py new file mode 100644 index 00000000..4b6512e4 --- /dev/null +++ b/tests/github/scripts/lib/test_inventory.py @@ -0,0 +1,35 @@ +import sys +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +sys.path.insert(0, str(REPO_ROOT / ".github/scripts")) + +from lib.inventory import render_inventory_markdown # noqa: E402 + + +def test_document_support_heading_is_vendor_neutral() -> None: + sections = { + "Instructions": [], + "Skills": [ + ".github/skills/anthropic-docx/SKILL.md", + ".github/skills/anthropic-pdf/SKILL.md", + ".github/skills/anthropic-pptx/SKILL.md", + ".github/skills/anthropic-xlsx/SKILL.md", + ], + "Scripts": [], + "Agents": [], + "Prompts": [], + } + + rendered = render_inventory_markdown(sections) + + assert "Support-only imported document skills" in rendered + assert "anthropic-docx" in rendered + assert "anthropic-pdf" in rendered + assert "anthropic-pptx" in rendered + assert "anthropic-xlsx" in rendered + assert "openai-* office skills" not in rendered diff --git a/tests/github/skills/grill-me/test_grill_me_metadata.py b/tests/github/skills/grill-me/test_grill_me_metadata.py new file mode 100644 index 00000000..22af67c3 --- /dev/null +++ b/tests/github/skills/grill-me/test_grill_me_metadata.py @@ -0,0 +1,17 @@ +from pathlib import Path + +import yaml + +SKILL_ROOT = Path(__file__).parents[4] / ".github" / "skills" / "grill-me" + + +def test_grill_me_metadata_identifies_the_interview_skill() -> None: + interface = yaml.safe_load( + (SKILL_ROOT / "agents" / "openai.yaml").read_text(encoding="utf-8") + )["interface"] + + assert interface["display_name"] == "Grill Me" + assert ( + interface["short_description"] + == "Relentless interview to sharpen a plan or design" + ) diff --git a/tests/github/skills/internal-agent-creator/test_agent_authoring_contract.py b/tests/github/skills/internal-agent-creator/test_agent_authoring_contract.py new file mode 100644 index 00000000..e92a160a --- /dev/null +++ b/tests/github/skills/internal-agent-creator/test_agent_authoring_contract.py @@ -0,0 +1,39 @@ +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[4] +SKILL_DIR = ROOT / ".github" / "skills" / "internal-agent-creator" + + +def read_bundle_file(relative_path: str) -> str: + return (SKILL_DIR / relative_path).read_text(encoding="utf-8") + + +def test_skill_routes_to_requirements_and_persona_reference() -> None: + skill = read_bundle_file("SKILL.md") + + assert "references/requirements-and-persona.md" in skill + assert "requirements gate" in skill.lower() + assert "context handoff" in skill.lower() + + +def test_requirements_and_persona_contract_is_safe_and_behavioral() -> None: + contract = read_bundle_file("references/requirements-and-persona.md") + + assert "## Requirements Gate" in contract + assert "## Persona Translation" in contract + assert "## Name and Path Safety" in contract + assert "## Context Handoff" in contract + assert "observable behavior" in contract + assert "Do not invent credentials" in contract + assert r"^internal-[a-z0-9]+(?:-[a-z0-9]+)*$" in contract + assert ".github/agents/" in contract + + +def test_templates_and_review_checklist_enforce_operating_stance() -> None: + template = read_bundle_file("references/agent-template.md") + checklist = read_bundle_file("references/review-checklist.md") + + assert "operating stance" in template.lower() + assert "prestige biography" in template.lower() + assert "requirements gate" in checklist.lower() + assert "context handoff" in checklist.lower() diff --git a/tests/github/skills/internal-aws/test_internal_aws_routing_contract.py b/tests/github/skills/internal-aws/test_internal_aws_routing_contract.py new file mode 100644 index 00000000..eda79e3e --- /dev/null +++ b/tests/github/skills/internal-aws/test_internal_aws_routing_contract.py @@ -0,0 +1,178 @@ +import re +from pathlib import Path + +import yaml + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SKILL_DIR = REPO_ROOT / ".github/skills/internal-aws" +LEGACY_SKILL_DIR = REPO_ROOT / ".github/skills" / ("internal-aws-" + "strategic") +SKILL_PATH = SKILL_DIR / "SKILL.md" +AGENT_PATH = SKILL_DIR / "agents/openai.yaml" +ROUTING_MATRIX_PATH = SKILL_DIR / "references/routing-matrix.md" + +EXPECTED_DESCRIPTION = ( + "Use when an AWS task cannot be routed confidently to a specific AWS " + "skill because the request is materially ambiguous, has multiple AWS domains " + "with no clear primary owner, or requires clarification before selecting the " + "correct specialist, or when the user needs high-level AWS platform " + "decision support or tradeoff framing before implementation. Do not use for " + "clearly scoped organization structure, governance or IAM, operations or " + "validation, Lambda, or current AWS documentation research tasks." +) + + +def load_frontmatter(path: Path) -> dict[str, object]: + _, raw_frontmatter, _ = path.read_text().split("---", maxsplit=2) + return yaml.safe_load(raw_frontmatter) + + +def test_internal_aws_replaces_the_legacy_bundle() -> None: + assert SKILL_PATH.is_file() + assert not LEGACY_SKILL_DIR.exists() + assert load_frontmatter(SKILL_PATH) == { + "name": "internal-aws", + "description": EXPECTED_DESCRIPTION, + } + + +def test_internal_aws_contract_is_router_and_strategic() -> None: + skill_text = SKILL_PATH.read_text() + + router_markers = ( + "material routing uncertainty", + "Do not activate only because the task concerns AWS", + "Do not activate when one specialist clearly owns the next step", + "Select the minimum specialist set", + "Explicit `$internal-aws` invocation remains valid", + ) + for marker in router_markers: + assert marker in skill_text + + strategic_markers = ( + "Identify the decision first, not the implementation tool", + "Compare realistic options, not strawmen.", + "Keep tradeoffs concrete.", + ) + for marker in strategic_markers: + assert marker in skill_text + + +def test_internal_aws_interface_names_router_and_strategic() -> None: + interface = yaml.safe_load(AGENT_PATH.read_text())["interface"] + + assert interface == { + "display_name": "Internal AWS", + "short_description": "AWS routing and strategic decision support", + "default_prompt": ( + "Use $internal-aws to route an unclear AWS task to the minimum " + "specialist set, or to frame an AWS decision when the next " + "step is not yet structure, governance, operations, or delivery." + ), + } + + +def test_routing_matrix_covers_positive_negative_and_multi_domain_cases() -> None: + matrix_text = ROUTING_MATRIX_PATH.read_text() + + for heading in ( + "## Fallback-positive cases", + "## Direct-specialist negative cases", + "## Multi-domain primary-owner cases", + "## Review rule", + ): + assert heading in matrix_text + + +AWS_SKILL_PATHS = sorted((REPO_ROOT / ".github/skills").glob("internal-aws*/SKILL.md")) +LEGACY_SKILL_ID = "internal-aws-" + "strategic" +FORBIDDEN_GENERIC_REFERENCES = ( + "internal-bash-script", + "internal-python-script", + "internal-python", + "internal-python-project", + "internal-nodejs", + "internal-nodejs-project", + "internal-terraform", +) + + +def test_aws_family_has_no_legacy_or_generic_skill_references() -> None: + assert len(AWS_SKILL_PATHS) == 6 + + for path in AWS_SKILL_PATHS: + skill_text = path.read_text() + assert LEGACY_SKILL_ID not in skill_text + for forbidden_name in FORBIDDEN_GENERIC_REFERENCES: + assert f"`{forbidden_name}`" not in skill_text + + +def test_specialists_name_internal_aws_only_as_uncertainty_fallback() -> None: + specialist_paths = [path for path in AWS_SKILL_PATHS if path != SKILL_PATH] + + for path in specialist_paths: + skill_text = path.read_text() + assert "`internal-aws`" in skill_text + assert "material routing uncertainty" in skill_text + + +LANE_SKILL_IDS = ( + "internal-aws-governance", + "internal-aws-lambda", + "internal-aws-mcp-research", + "internal-aws-operations", + "internal-aws-organization-structure", +) + +SKILL_REFERENCE_PATTERN = re.compile( + r"`((?:internal|awesome|openai|superpowers|agent-os|antigravity|addyosmani" + r"|local|mattpocock|terraform|vercel|customize|grill|graphify)-[a-z0-9-]+)`" +) + +REMOVED_SECTION_HEADINGS = ( + "## Handoffs", + "## Cross-references", + "## Referenced skills", + "## Relationship to adjacent skills", + "## When not to use", +) + + +def test_lane_skills_have_no_sibling_references_or_handoffs() -> None: + for skill_id in LANE_SKILL_IDS: + skill_dir = REPO_ROOT / ".github/skills" / skill_id + text_paths = [ + skill_dir / "SKILL.md", + *sorted(skill_dir.glob("references/*.md")), + ] + for text_path in text_paths: + skill_text = text_path.read_text() + for heading in REMOVED_SECTION_HEADINGS: + assert heading not in skill_text, f"{skill_id} keeps {heading}" + references = set(SKILL_REFERENCE_PATTERN.findall(skill_text)) + assert references <= {"internal-aws"}, ( + f"{text_path.name} references {sorted(references)}" + ) + assert "handoff" not in skill_text.lower(), ( + f"{text_path.name} still mentions handoffs" + ) + + +def test_mcp_capability_map_uses_the_canonical_fallback_name() -> None: + capability_map = ( + REPO_ROOT + / ".github/skills/internal-aws-mcp-research/references/mcp-capabilities.md" + ).read_text() + + assert LEGACY_SKILL_ID not in capability_map + assert "`internal-aws`" in capability_map + + +def test_inventory_lists_only_the_canonical_internal_aws_bundle() -> None: + inventory_text = (REPO_ROOT / ".github/INVENTORY.md").read_text() + + assert ".github/skills/internal-aws/SKILL.md" in inventory_text + assert LEGACY_SKILL_ID not in inventory_text diff --git a/tests/github/skills/internal-azure/test_internal_azure_routing_contract.py b/tests/github/skills/internal-azure/test_internal_azure_routing_contract.py new file mode 100644 index 00000000..4aee4676 --- /dev/null +++ b/tests/github/skills/internal-azure/test_internal_azure_routing_contract.py @@ -0,0 +1,207 @@ +import re +from pathlib import Path + +import yaml + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SKILL_DIR = REPO_ROOT / ".github/skills/internal-azure" +LEGACY_SKILL_DIR = REPO_ROOT / ".github/skills" / ("internal-azure-" + "strategic") +SKILL_PATH = SKILL_DIR / "SKILL.md" +AGENT_PATH = SKILL_DIR / "agents/openai.yaml" +ROUTING_MATRIX_PATH = SKILL_DIR / "references/routing-matrix.md" +LENS_PLAYBOOK_PATH = SKILL_DIR / "references/lens-playbook.md" + +EXPECTED_DESCRIPTION = ( + "Use when an Azure task cannot be routed confidently to a specific Azure " + "skill because the request is materially ambiguous, has multiple Azure " + "domains with no clear primary owner, or requires clarification before " + "selecting the correct specialist, or when the user needs high-level Azure " + "platform decision support or tradeoff framing before implementation. " + "Do not use for clearly scoped organization structure, governance or " + "identity, operations or validation, or Azure DevOps pipeline tasks." +) + + +def load_frontmatter(path: Path) -> dict[str, object]: + _, raw_frontmatter, _ = path.read_text().split("---", maxsplit=2) + return yaml.safe_load(raw_frontmatter) + + +def test_internal_azure_replaces_the_legacy_bundle() -> None: + assert SKILL_PATH.is_file() + assert not LEGACY_SKILL_DIR.exists() + assert load_frontmatter(SKILL_PATH) == { + "name": "internal-azure", + "description": EXPECTED_DESCRIPTION, + } + + +def test_internal_azure_contract_is_router_and_strategic() -> None: + skill_text = SKILL_PATH.read_text() + + router_markers = ( + "material routing uncertainty", + "Do not activate only because the task concerns Azure", + "Do not activate when one specialist clearly owns the next step", + "Select the minimum specialist set", + "Explicit `$internal-azure` invocation remains valid", + ) + for marker in router_markers: + assert marker in skill_text + + strategic_markers = ( + "Identify the decision first, not the implementation tool", + "Compare realistic options, not strawmen.", + "Keep tradeoffs concrete.", + ) + for marker in strategic_markers: + assert marker in skill_text + + +def test_internal_azure_interface_names_router_and_strategic() -> None: + interface = yaml.safe_load(AGENT_PATH.read_text())["interface"] + + assert interface == { + "display_name": "Internal Azure", + "short_description": "Azure routing and strategic decision support", + "default_prompt": ( + "Use $internal-azure to route an unclear Azure task to the minimum " + "specialist set, or to frame an Azure decision when the next " + "step is not yet structure, governance, operations, or delivery." + ), + } + + +def test_routing_matrix_covers_positive_negative_and_multi_domain_cases() -> None: + matrix_text = ROUTING_MATRIX_PATH.read_text() + + for heading in ( + "## Fallback-positive cases", + "## Direct-specialist negative cases", + "## Multi-domain primary-owner cases", + "## Review rule", + ): + assert heading in matrix_text + + +def test_lens_playbook_keeps_strategic_depth() -> None: + playbook_text = LENS_PLAYBOOK_PATH.read_text() + + for heading in ( + "## Common lens combinations", + "## Decision note pattern", + "## Depth control", + ): + assert heading in playbook_text + + +AZURE_SKILL_PATHS = sorted( + (REPO_ROOT / ".github/skills").glob("internal-azure*/SKILL.md") +) +LEGACY_SKILL_ID = "internal-azure-" + "strategic" +FORBIDDEN_GENERIC_REFERENCES = ( + "internal-bash-script", + "internal-python-script", + "internal-python", + "internal-python-project", + "internal-nodejs", + "internal-nodejs-project", + "internal-terraform", +) + +EXPECTED_SPECIALIST_DESCRIPTION_PREFIXES = { + "internal-azure-organization-structure": "Use when ", + "internal-azure-governance": "Use when ", + "internal-azure-operations": "Use when ", + "internal-azure-devops": "Use when ", +} + + +def test_azure_family_has_no_legacy_or_generic_skill_references() -> None: + assert len(AZURE_SKILL_PATHS) == 5 + + for path in AZURE_SKILL_PATHS: + skill_text = path.read_text() + assert LEGACY_SKILL_ID not in skill_text + for forbidden_name in FORBIDDEN_GENERIC_REFERENCES: + assert f"`{forbidden_name}`" not in skill_text + + +def test_specialists_name_internal_azure_only_as_uncertainty_fallback() -> None: + specialist_paths = [path for path in AZURE_SKILL_PATHS if path != SKILL_PATH] + + for path in specialist_paths: + skill_text = path.read_text() + assert "`internal-azure`" in skill_text + assert "material routing uncertainty" in skill_text + + +LANE_SKILL_IDS = ( + "internal-azure-governance", + "internal-azure-operations", + "internal-azure-organization-structure", +) + +SKILL_REFERENCE_PATTERN = re.compile( + r"`((?:internal|awesome|openai|superpowers|agent-os|antigravity|addyosmani" + r"|local|mattpocock|terraform|vercel|customize|grill|graphify)-[a-z0-9-]+)`" +) + +REMOVED_SECTION_HEADINGS = ( + "## Handoffs", + "## Cross-references", + "## Referenced skills", + "## Relationship to adjacent skills", + "## When not to use", +) + + +def test_lane_skills_have_no_sibling_references_or_handoffs() -> None: + for skill_id in LANE_SKILL_IDS: + skill_dir = REPO_ROOT / ".github/skills" / skill_id + text_paths = [ + skill_dir / "SKILL.md", + *sorted(skill_dir.glob("references/*.md")), + ] + for text_path in text_paths: + skill_text = text_path.read_text() + for heading in REMOVED_SECTION_HEADINGS: + assert heading not in skill_text, f"{skill_id} keeps {heading}" + references = set(SKILL_REFERENCE_PATTERN.findall(skill_text)) + assert references <= {"internal-azure"}, ( + f"{text_path.name} references {sorted(references)}" + ) + assert "handoff" not in skill_text.lower(), ( + f"{text_path.name} still mentions handoffs" + ) + + +def test_specialist_descriptions_carry_positive_and_negative_triggers() -> None: + for path in AZURE_SKILL_PATHS: + frontmatter = load_frontmatter(path) + name = frontmatter["name"] + if name == "internal-azure": + continue + assert name in EXPECTED_SPECIALIST_DESCRIPTION_PREFIXES + description = frontmatter["description"] + assert description.startswith(EXPECTED_SPECIALIST_DESCRIPTION_PREFIXES[name]) + assert "Do not use" in description + + +def test_inventory_lists_only_the_canonical_internal_azure_bundle() -> None: + inventory_text = (REPO_ROOT / ".github/INVENTORY.md").read_text() + + assert ".github/skills/internal-azure/SKILL.md" in inventory_text + assert LEGACY_SKILL_ID not in inventory_text + + +def test_repo_profiles_reference_the_canonical_internal_azure_bundle() -> None: + profiles = yaml.safe_load((REPO_ROOT / ".github/repo-profiles.yml").read_text()) + azure_skills = profiles["profiles"]["azure-platform"]["recommended_skills"] + + assert "skills/internal-azure/SKILL.md" in azure_skills + assert not any(LEGACY_SKILL_ID in entry for entry in azure_skills) diff --git a/tests/github/skills/internal-excel/test_internal_excel_contract.py b/tests/github/skills/internal-excel/test_internal_excel_contract.py new file mode 100644 index 00000000..be02763f --- /dev/null +++ b/tests/github/skills/internal-excel/test_internal_excel_contract.py @@ -0,0 +1,21 @@ +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SKILL_PATH = REPO_ROOT / ".github/skills/internal-excel/SKILL.md" +TOOL_SELECTION_PATH = ( + REPO_ROOT / ".github/skills/internal-excel/references/tool-selection.md" +) + + +def test_internal_excel_routes_workbook_presentation_to_anthropic_xlsx() -> None: + skill_text = SKILL_PATH.read_text() + tool_selection_text = TOOL_SELECTION_PATH.read_text() + + assert "anthropic-xlsx" in skill_text + assert "anthropic-xlsx" in tool_selection_text + assert "openai-spreadsheet" not in skill_text + assert "openai-spreadsheet" not in tool_selection_text diff --git a/tests/github/skills/internal-gateway-codebase-improvement/test_contract.py b/tests/github/skills/internal-gateway-codebase-improvement/test_contract.py new file mode 100644 index 00000000..8b9cca92 --- /dev/null +++ b/tests/github/skills/internal-gateway-codebase-improvement/test_contract.py @@ -0,0 +1,99 @@ +from pathlib import Path + +import yaml + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SKILL_ROOT = REPO_ROOT / ".github/skills/internal-gateway-codebase-improvement" +SKILL_PATH = SKILL_ROOT / "SKILL.md" +OPENAI_PATH = SKILL_ROOT / "agents/openai.yaml" +CONTRACT_PATH = REPO_ROOT / "INTERNAL_CONTRACT.md" +AGENT_WRAPPER = ( + REPO_ROOT / ".github/agents/internal-gateway-codebase-improvement.agent.md" +) + + +def _frontmatter() -> dict[str, object]: + text = SKILL_PATH.read_text(encoding="utf-8") + return yaml.safe_load(text.split("---", 2)[1]) + + +def test_skill_is_manual_only_in_both_metadata_surfaces() -> None: + assert _frontmatter()["name"] == "internal-gateway-codebase-improvement" + assert _frontmatter()["disable-model-invocation"] is True + policy = yaml.safe_load(OPENAI_PATH.read_text(encoding="utf-8"))["policy"] + assert policy["allow_implicit_invocation"] is False + + +def test_skill_has_no_agent_wrapper() -> None: + assert not AGENT_WRAPPER.exists() + + +def test_contract_classifies_the_skill_outside_the_gateway_core() -> None: + contract = CONTRACT_PATH.read_text(encoding="utf-8") + required = ( + "`internal-gateway-codebase-improvement` is a manual-only " + "specialist gateway outside the canonical gateway core" + ) + assert required in " ".join(contract.split()) + assert "must not become an implicit fallback" in contract + + +WORKFLOW_PATH = SKILL_ROOT / "references/workflow.md" +LANES = ( + "local-simplification", + "architecture-improvement", + "combined", +) +METHOD_OWNERS = ( + "internal-tdd", + "superpowers-verification-before-completion", +) +REFERENCE_ONLY_OWNERS = ( + "mattpocock-improve-codebase-architecture", + "addyosmani-code-simplification", +) + + +def test_skill_defines_exactly_three_evidence_selected_lanes() -> None: + skill = SKILL_PATH.read_text(encoding="utf-8") + workflow = WORKFLOW_PATH.read_text(encoding="utf-8") + for lane in LANES: + assert f"`{lane}`" in skill + assert lane in workflow + assert "Select exactly one lane from repository evidence" in skill + assert "Do not run both source methods by default" in skill + assert "No silent lane escalation" in skill + + +def test_skill_references_existing_method_owners_without_copying_them() -> None: + skill = SKILL_PATH.read_text(encoding="utf-8") + for owner in METHOD_OWNERS: + assert f"`/{owner}`" in skill + for owner in REFERENCE_ONLY_OWNERS: + assert f"`/{owner}`" in skill + assert "HTML report scaffold" not in skill + assert "The Five Principles" not in skill + + +def test_structural_work_requires_approval_and_protects_seams() -> None: + skill = SKILL_PATH.read_text(encoding="utf-8") + workflow = WORKFLOW_PATH.read_text(encoding="utf-8") + required = ( + "Structural Approval Gate", + "Protected seam set", + "Passing behavior baseline", + "Final Evidence Gate", + ) + for marker in required: + assert marker in skill + assert marker in workflow + assert workflow.index("Structural Approval Gate") < workflow.index( + "Executable refactor" + ) + assert workflow.index("Protected seam set") < workflow.index( + "Behavior-preserving simplification" + ) diff --git a/tests/github/skills/internal-gateway-critical-master/scripts/test_critical_master_helpers.py b/tests/github/skills/internal-gateway-critical-master/scripts/test_critical_master_helpers.py index 144c27c9..43afaca1 100644 --- a/tests/github/skills/internal-gateway-critical-master/scripts/test_critical_master_helpers.py +++ b/tests/github/skills/internal-gateway-critical-master/scripts/test_critical_master_helpers.py @@ -26,14 +26,25 @@ def test_count_words_ignores_fenced_code_blocks() -> None: assert critical_master.count_words("alpha ```python\nbeta\n``` gamma") == 2 -def test_parse_findings_detects_optional_root_question() -> None: - findings = critical_master.parse_findings( - "### 1. Audit trail weakens\n\n" - "- **Impact:** Central logs become incomplete.\n" - "- **Evidence:** `inference` - no replacement is described.\n" - "- **Mitigation:** Add a signed attestation.\n" - "- **Question:** Which audit record replaces CI?\n" +def test_parse_critical_card_uses_emoji_as_language_neutral_keys() -> None: + card = critical_master.parse_critical_card( + "🎯 **Piano:** Spostare i controlli sui computer locali.\n" + "⚠️ **Critica:** Perderemmo la prova centrale perché i controlli locali " + "non producono un registro condiviso.\n" + "✅ **Consiglio:** Mantenere la CI fino a un sostituto centrale.\n" ) - assert len(findings) == 1 - assert findings[0].has_question + assert tuple(card.by_marker) == ("🎯", "⚠️", "✅") + assert card.by_marker["⚠️"].content.startswith("Perderemmo") + + +def test_parse_critical_card_tracks_optional_risk_and_question() -> None: + card = critical_master.parse_critical_card( + "🎯 **Plan:** Move validation to developer machines.\n" + "⚠️ **Critique:** Central proof disappears because local checks are private.\n" + "💥 **Risk:** Some repositories may silently skip validation.\n" + "✅ **Advice:** Keep CI until an equivalent central control exists.\n" + "❓ **Open point:** What officially replaces the CI logs?\n" + ) + + assert tuple(card.by_marker) == ("🎯", "⚠️", "💥", "✅", "❓") diff --git a/tests/github/skills/internal-gateway-critical-master/scripts/test_critical_master_validator.py b/tests/github/skills/internal-gateway-critical-master/scripts/test_critical_master_validator.py index ae4845f8..d20b5f79 100644 --- a/tests/github/skills/internal-gateway-critical-master/scripts/test_critical_master_validator.py +++ b/tests/github/skills/internal-gateway-critical-master/scripts/test_critical_master_validator.py @@ -30,116 +30,226 @@ def run_validator(*args: str) -> subprocess.CompletedProcess[str]: ) -def test_valid_fixture_passes() -> None: - result = run_validator("--file", str(FIXTURES / "critical_output_valid.md")) +def test_valid_minimal_card_passes_strict() -> None: + result = run_validator( + "--file", str(FIXTURES / "critical_output_valid.md"), "--strict" + ) assert result.returncode == 0 assert result.stdout.strip() == "" -def test_invalid_fixture_fails() -> None: +def test_valid_complex_card_passes_strict() -> None: result = run_validator( "--file", - str(FIXTURES / "critical_output_invalid_missing_section.md"), + str(FIXTURES / "critical_output_valid_premortem.md"), + "--strict", ) - assert result.returncode != 0 - assert "Required section" in result.stdout + assert result.returncode == 0 -def test_strict_mode_fails_on_advisory_finding() -> None: - result = run_validator( - "--file", - str(FIXTURES / "critical_output_advisory.md"), - "--strict", +def test_legacy_section_report_is_rejected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "## Summary\n\nOld summary.\n\n" + "## Findings\n\nOld findings.\n\n" + "## Synthesis\n\nOld synthesis.\n\n" + "## Outcome\n\n`accept-with-risk`\n" ) - assert result.returncode != 0 - assert "summary-word-limit" in result.stdout or "total-word-limit" in result.stdout + assert "legacy-section-format" in {finding.code for finding in findings} + + +def test_missing_plan_marker_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "⚠️ **Critique:** Something is wrong.\n✅ **Advice:** Do this instead.\n" + ) + assert "missing-plan" in {finding.code for finding in findings} + + +def test_missing_critique_marker_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Plan:** Do something.\n✅ **Advice:** Do this instead.\n" + ) + assert "missing-critique" in {finding.code for finding in findings} -def test_question_word_limit_is_advisory() -> None: - text = """ -## Summary +def test_missing_advice_marker_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Plan:** Do something.\n⚠️ **Critique:** Something is wrong.\n" + ) + assert "missing-advice" in {finding.code for finding in findings} -We are challenging whether local validation can replace CI validation. -## Findings +def test_duplicate_marker_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Plan:** First plan.\n" + "🎯 **Plan:** Duplicate plan.\n" + "⚠️ **Critique:** Something is wrong.\n" + "✅ **Advice:** Do this instead.\n" + ) + assert "duplicate-marker" in {finding.code for finding in findings} + + +def test_incorrect_marker_order_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "⚠️ **Critique:** Critique before plan.\n" + "🎯 **Plan:** Plan after critique.\n" + "✅ **Advice:** Do this instead.\n" + ) + assert "card-line-order" in {finding.code for finding in findings} + + +def test_risk_after_advice_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Plan:** Do something.\n" + "⚠️ **Critique:** Something is wrong.\n" + "✅ **Advice:** Do this instead.\n" + "💥 **Risk:** Material risk.\n" + ) + assert "card-line-order" in {finding.code for finding in findings} + + +def test_question_before_advice_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Plan:** Do something.\n" + "⚠️ **Critique:** Something is wrong.\n" + "❓ **Question:** What if?\n" + "✅ **Advice:** Do this instead.\n" + ) + assert "card-line-order" in {finding.code for finding in findings} -### 1. The audit trail weakens -- **Impact:** Central CI logs become incomplete. -- **Evidence:** `inference` - no replacement logging is described. -- **Mitigation:** Define a durable audit record before replacing CI. -- **Question:** Which durable centrally searchable independently retained signed audit record replaces the CI validation log for reviewers, compliance checks, later investigations, audit replay, governance reporting, incident review, and rollout approval? +def test_more_than_five_lines_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Plan:** Do something.\n" + "⚠️ **Critique:** Something is wrong.\n" + "💥 **Risk:** Material risk.\n" + "💥 **Risk:** Another risk.\n" + "✅ **Advice:** Do this instead.\n" + "❓ **Question:** What if?\n" + ) + assert any( + finding.code in ("card-line-count", "duplicate-marker") for finding in findings + ) -## Synthesis -The strongest risk is compliance visibility. +def test_unexpected_content_line_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Plan:** Do something.\n" + "⚠️ **Critique:** Something is wrong.\n" + "✅ **Advice:** Do this instead.\n" + "Some random prose line.\n" + ) + assert "unexpected-content-line" in {finding.code for finding in findings} -## Outcome -`accept-with-risk` -""" +def test_empty_content_after_label_detected() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Plan:**\n" + "⚠️ **Critique:** Something is wrong.\n" + "✅ **Advice:** Do this instead.\n" + ) + assert "unexpected-content-line" in {finding.code for finding in findings} - findings = VALIDATOR_MODULE.validate_output(text) - assert any(finding.code == "finding-question-word-limit" for finding in findings) +def test_per_line_word_budget_enforced() -> None: + long_line = "word " * 35 + findings = VALIDATOR_MODULE.validate_output( + f"🎯 **Plan:** {long_line.strip()}\n" + "⚠️ **Critique:** Short critique.\n" + "✅ **Advice:** Short advice.\n" + ) + assert "card-line-word-limit" in {finding.code for finding in findings} -def test_objection_word_limit_is_advisory() -> None: - long_objection = " ".join( - [ - "This", - "objection", - "heading", - "intentionally", - "exceeds", - "the", - "thirty", - "word", - "limit", - "by", - "repeating", - "core", - "concerns", - "about", - "the", - "proposal", - "without", - "adding", - "any", - "new", - "signal", - "for", - "the", - "reader", - "and", - "must", - "be", - "shortened", - "now", - "again", - "finally", - ] +def test_total_word_budget_enforced() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Plan:** " + "word " * 35 + "\n" + "⚠️ **Critique:** " + "word " * 35 + "\n" + "✅ **Advice:** " + "word " * 35 + "\n" ) - text = f""" -## Summary + assert "total-word-limit" in {finding.code for finding in findings} -We are testing the objection word-limit enforcement. -## Findings +def test_localized_labels_accepted_when_emoji_order_valid() -> None: + findings = VALIDATOR_MODULE.validate_output( + "🎯 **Piano:** Spostare la validazione.\n" + "⚠️ **Critica:** La prova centrale scompare.\n" + "✅ **Consiglio:** Mantenere la CI.\n" + ) + blocking_codes = {f.code for f in findings if f.severity == "blocking"} + assert not blocking_codes -### 1. {long_objection} -- **Impact:** Scope ambiguity risks rejection. -- **Evidence:** `inference` — no attachment to contract. -- **Mitigation:** Tighten scope before approval. +def test_cli_format_text_renders_findings() -> None: + text = ( + "🎯 **Plan:** Do something.\n" + "⚠️ **Critique:** Something is wrong.\n" + "Some stray line.\n" + ) + result = subprocess.run( + [sys.executable, str(VALIDATOR), "--format", "text"], + input=text, + cwd=REPO_ROOT, + text=True, + capture_output=True, + check=False, + ) + assert result.returncode == 1 + assert "[BLOCKING]" in result.stdout + assert "unexpected-content-line" in result.stdout + + +def test_cli_format_json_returns_finding_list() -> None: + import json as json_mod + + result = subprocess.run( + [sys.executable, str(VALIDATOR), "--format", "json"], + input=( + "🎯 **Plan:** Do something.\n" + "⚠️ **Critique:** Something is wrong.\n" + "✅ **Advice:** Do this.\n" + "Extra line.\n" + ), + cwd=REPO_ROOT, + text=True, + capture_output=True, + check=False, + ) + data = json_mod.loads(result.stdout) + assert isinstance(data, list) + assert len(data) > 0 + assert "code" in data[0] + + +def test_cli_format_compact_returns_status_and_counts() -> None: + import json as json_mod + + result = subprocess.run( + [sys.executable, str(VALIDATOR), "--format", "compact"], + input=( + "🎯 **Plan:** Do something.\n" + "⚠️ **Critique:** Something is wrong.\n" + "✅ **Advice:** Do this.\n" + ), + cwd=REPO_ROOT, + text=True, + capture_output=True, + check=False, + ) + data = json_mod.loads(result.stdout) + assert "status" in data + assert "finding_counts" in data -## Outcome -`accept-with-risk` +def test_cli_unreadable_file_exits_nonzero_with_stderr() -> None: + result = run_validator( + "--file", + "/nonexistent/path/missing_file.md", + ) + assert result.returncode != 0 + assert result.stderr.strip() != "" -## Synthesis -The challenge surfaces one open question. -""" - findings = VALIDATOR_MODULE.validate_output(text) - assert any(finding.code == "finding-objection-word-limit" for finding in findings) +def test_cli_make_target_includes_strict() -> None: + makefile = (REPO_ROOT / "Makefile").read_text(encoding="utf-8") + assert "--strict" in makefile + assert "critical-validate" in makefile diff --git a/tests/github/skills/internal-gateway-critical-master/test_contract_alignment.py b/tests/github/skills/internal-gateway-critical-master/test_contract_alignment.py new file mode 100644 index 00000000..97c835c9 --- /dev/null +++ b/tests/github/skills/internal-gateway-critical-master/test_contract_alignment.py @@ -0,0 +1,152 @@ +import re +import sys +from importlib.util import module_from_spec, spec_from_file_location +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) + +BUNDLE = REPO_ROOT / ".github/skills/internal-gateway-critical-master" + +MODULE_PATH = BUNDLE / "scripts/critical_master.py" +MODULE_SPEC = spec_from_file_location("critical_master_cm", MODULE_PATH) +assert MODULE_SPEC is not None and MODULE_SPEC.loader is not None +critical_master = module_from_spec(MODULE_SPEC) +sys.modules[MODULE_SPEC.name] = critical_master +MODULE_SPEC.loader.exec_module(critical_master) + + +def _load_text(relative_path: str) -> str: + return (BUNDLE / relative_path).read_text(encoding="utf-8") + + +SKILL_TEXT = _load_text("SKILL.md") +CONTRACT_TEXT = _load_text("references/output-contract.md") +AGENT_YAML = _load_text("agents/openai.yaml") +FIXTURE_TEXT = _load_text("fixtures/critical_output_valid.md") +AGENT_MD_TEXT = ( + REPO_ROOT / ".github/agents/internal-gateway-critical-master.agent.md" +).read_text(encoding="utf-8") + +EXPECTED_OUTCOMES = { + "reformulate-plan", + "de-escalate-to-simple", + "route-to-execution-owner", + "review-evidence", + "continue-critical-with-new-evidence", + "accept-with-risk", +} + + +def test_skill_requires_exactly_three_lenses() -> None: + assert "Select exactly **three lenses**" in SKILL_TEXT + + +def test_skill_requires_lateral_third_lens() -> None: + assert "lens three must be lateral" in SKILL_TEXT.lower() + + +def test_skill_keeps_full_analysis_internal_by_default() -> None: + assert "internal critical record" in SKILL_TEXT.lower() + assert "do not print" in SKILL_TEXT.lower() + + +def test_contract_requires_adaptive_card_markers() -> None: + for marker in ("🎯", "⚠️", "✅"): + assert marker in CONTRACT_TEXT + assert "three to five" in CONTRACT_TEXT.lower() + + +def test_contract_marks_risk_and_question_optional() -> None: + assert "💥" in CONTRACT_TEXT + assert "❓" in CONTRACT_TEXT + assert "only" in CONTRACT_TEXT.lower() + + +def test_agent_prompt_requests_compact_localized_projection() -> None: + assert "three-to-five-line" in AGENT_YAML + assert "user's language" in AGENT_YAML + assert "600 words" not in AGENT_YAML + + +def test_valid_fixture_has_no_legacy_sections() -> None: + assert "## Summary" not in FIXTURE_TEXT + assert "## Findings" not in FIXTURE_TEXT + + +def test_agent_yaml_does_not_duplicate_lens_count() -> None: + assert "2-3 lenses" not in AGENT_YAML + + +def test_allowed_outcomes_match_expected_set() -> None: + assert critical_master.ALLOWED_OUTCOMES == frozenset(EXPECTED_OUTCOMES) + + +def test_skill_contains_route_to_execution_owner() -> None: + assert "`route-to-execution-owner`" in SKILL_TEXT + + +def test_skill_contains_continue_critical_with_new_evidence() -> None: + assert "`continue-critical-with-new-evidence`" in SKILL_TEXT + + +def test_skill_outcome_table_matches_allowed_outcomes() -> None: + table_rows = re.findall(r"^\|\s*`([^`]+)`\s*\|", SKILL_TEXT, re.MULTILINE) + outcome_values = { + row.strip() + for row in table_rows + if row.strip() + in { + "reformulate-plan", + "de-escalate-to-simple", + "route-to-execution-owner", + "review-evidence", + "continue-critical-with-new-evidence", + "accept-with-risk", + "execute-clear-next-step", + "continue-critical", + } + } + assert outcome_values == EXPECTED_OUTCOMES + + +def test_agent_boundary_keeps_internal_record_hidden() -> None: + assert "internal" in AGENT_MD_TEXT.lower() + assert "card" in AGENT_MD_TEXT.lower() + + +def _load_routing_cases() -> list[dict]: + import json + + path = BUNDLE / "fixtures/routing_cases.json" + return json.loads(path.read_text(encoding="utf-8")) + + +def test_routing_case_ids_are_unique() -> None: + cases = _load_routing_cases() + ids = [c["id"] for c in cases] + assert len(ids) == len(set(ids)) + + +def test_routing_cases_have_non_empty_prompts() -> None: + cases = _load_routing_cases() + for case in cases: + assert case["prompt"].strip() + + +def test_routing_cases_cover_expected_owners() -> None: + cases = _load_routing_cases() + owners = {c["expected_owner"] for c in cases} + assert "internal-gateway-critical-master" in owners + assert "internal-gateway-idea" in owners + assert "internal-gateway-simple-task" in owners + + +def test_critical_master_description_leads_with_challenge() -> None: + frontmatter_match = re.search(r"^description:\s*(.+)$", SKILL_TEXT, re.MULTILINE) + assert frontmatter_match is not None + description = frontmatter_match.group(1).strip().lower() + assert "critical challenge" in description or "critical" in description diff --git a/tests/github/skills/internal-gateway-execute-plans/scripts/conftest.py b/tests/github/skills/internal-gateway-execute-plans/scripts/conftest.py new file mode 100644 index 00000000..11a23de6 --- /dev/null +++ b/tests/github/skills/internal-gateway-execute-plans/scripts/conftest.py @@ -0,0 +1,33 @@ +import shutil +from pathlib import Path + +import pytest + +BUNDLE = ( + Path(__file__).resolve().parents[5] + / ".github/skills/internal-gateway-execute-plans" +) +FIXTURES = BUNDLE / "fixtures" + + +@pytest.fixture() +def valid_plan(tmp_path: Path) -> Path: + (tmp_path / "AGENTS.md").write_text("# agents\n") + (tmp_path / ".github").mkdir() + staged = tmp_path / "tmp" / "superpowers" / "plans" + staged.mkdir(parents=True) + target = staged / "valid-plan.md" + shutil.copy(FIXTURES / "valid-plan.md", target) + return target + + +@pytest.fixture() +def valid_partial_status(tmp_path: Path) -> Path: + target = tmp_path / "valid-plan.PARTIAL.md" + shutil.copy(FIXTURES / "valid-plan.PARTIAL.md", target) + return target + + +@pytest.fixture() +def invalid_status() -> Path: + return FIXTURES / "invalid-status.md" diff --git a/tests/github/skills/internal-gateway-execute-plans/scripts/test_plan_execution.py b/tests/github/skills/internal-gateway-execute-plans/scripts/test_plan_execution.py new file mode 100644 index 00000000..6475e19c --- /dev/null +++ b/tests/github/skills/internal-gateway-execute-plans/scripts/test_plan_execution.py @@ -0,0 +1,177 @@ +import json +import subprocess +import sys +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +BUNDLE = REPO_ROOT / ".github/skills/internal-gateway-execute-plans" +SCRIPTS = BUNDLE / "scripts" +FIXTURES = BUNDLE / "fixtures" + +sys.path.insert(0, str(SCRIPTS)) + +from plan_execution import ( # noqa: E402 + Finding, + build_compact_payload, + compute_sha256, + validate_plan, + validate_resume, + validate_status, +) + + +def _fixture(name: str) -> Path: + return FIXTURES / name + + +def test_valid_plan_is_bound_to_its_sha256(valid_plan: Path) -> None: + findings = validate_plan(valid_plan, repo_root=valid_plan.parents[3]) + assert findings == [] + assert compute_sha256(valid_plan).startswith("sha256:") + + +def test_plan_outside_retained_plan_directory_is_rejected(tmp_path: Path) -> None: + plan = tmp_path / "plan.md" + plan.write_text("# Plan\n") + findings = validate_plan(plan, repo_root=tmp_path) + assert {item.code for item in findings} >= {"plan-outside-retained-directory"} + + +def test_status_rejects_unknown_state_and_missing_headings( + invalid_status: Path, +) -> None: + findings = validate_status(invalid_status) + codes = {item.code for item in findings} + assert "unknown-status" in codes + assert "missing-heading" in codes + + +def test_resume_rejects_plan_fingerprint_drift( + valid_plan: Path, valid_partial_status: Path +) -> None: + valid_plan.write_text(valid_plan.read_text() + "\nChanged after approval.\n") + findings = validate_resume(valid_plan, valid_partial_status) + assert {item.code for item in findings} >= {"plan-fingerprint-drift"} + + +def test_compact_output_is_bounded() -> None: + payload = build_compact_payload([Finding("missing-heading", "detail", "blocking")]) + assert payload == { + "status": "failed", + "finding_counts": {"total": 1, "blocking": 1, "notice": 0}, + "finding_sample": [{"code": "missing-heading", "severity": "blocking"}], + "next_action": "Resolve blocking plan execution findings.", + } + + +def test_preflight_cli_valid_fixture(tmp_path: Path) -> None: + (tmp_path / "AGENTS.md").write_text("# agents\n") + (tmp_path / ".github").mkdir() + staged = tmp_path / "tmp" / "superpowers" / "plans" + staged.mkdir(parents=True) + plan = staged / "valid-plan.md" + plan.write_text(_fixture("valid-plan.md").read_text()) + repo_root = tmp_path + result = subprocess.run( + [ + sys.executable, + str(SCRIPTS / "plan_execution.py"), + "preflight", + str(plan), + "--repo-root", + str(repo_root), + "--format", + "compact", + ], + capture_output=True, + text=True, + ) + assert result.returncode == 0, result.stderr + + +def test_preflight_cli_rejects_plan_outside_directory() -> None: + result = subprocess.run( + [ + sys.executable, + str(SCRIPTS / "plan_execution.py"), + "preflight", + str(_fixture("valid-plan.md")), + "--repo-root", + str(REPO_ROOT), + "--format", + "compact", + ], + capture_output=True, + text=True, + ) + assert result.returncode != 0 + payload = json.loads(result.stdout) + assert payload["status"] == "failed" + + +def test_status_check_cli_invalid() -> None: + result = subprocess.run( + [ + sys.executable, + str(SCRIPTS / "plan_execution.py"), + "status-check", + str(_fixture("invalid-status.md")), + "--format", + "compact", + ], + capture_output=True, + text=True, + ) + assert result.returncode != 0 + payload = json.loads(result.stdout) + codes = {f["code"] for f in payload["finding_sample"]} + assert "unknown-status" in codes + + +def test_status_check_cli_valid() -> None: + result = subprocess.run( + [ + sys.executable, + str(SCRIPTS / "plan_execution.py"), + "status-check", + str(_fixture("valid-plan.PARTIAL.md")), + "--format", + "compact", + ], + capture_output=True, + text=True, + ) + assert result.returncode == 0, result.stderr + + +def test_resume_check_cli_detects_drift(tmp_path: Path) -> None: + (tmp_path / "AGENTS.md").write_text("# agents\n") + (tmp_path / ".github").mkdir() + staged = tmp_path / "tmp" / "superpowers" / "plans" + staged.mkdir(parents=True) + plan = staged / "valid-plan.md" + plan.write_text(_fixture("valid-plan.md").read_text() + "\nDrifted.\n") + status = tmp_path / "valid-plan.PARTIAL.md" + status.write_text(_fixture("valid-plan.PARTIAL.md").read_text()) + result = subprocess.run( + [ + sys.executable, + str(SCRIPTS / "plan_execution.py"), + "resume-check", + str(plan), + str(status), + "--format", + "compact", + ], + capture_output=True, + text=True, + cwd=str(tmp_path), + ) + assert result.returncode != 0 + payload = json.loads(result.stdout) + codes = {f["code"] for f in payload["finding_sample"]} + assert "plan-fingerprint-drift" in codes diff --git a/tests/github/skills/internal-gateway-execute-plans/test_execute_plans_simplification_routing_contract.py b/tests/github/skills/internal-gateway-execute-plans/test_execute_plans_simplification_routing_contract.py deleted file mode 100644 index 696413c2..00000000 --- a/tests/github/skills/internal-gateway-execute-plans/test_execute_plans_simplification_routing_contract.py +++ /dev/null @@ -1,26 +0,0 @@ -from pathlib import Path - -REPO_ROOT = next( - parent - for parent in Path(__file__).resolve().parents - if (parent / "AGENTS.md").exists() and (parent / ".github").exists() -) -SKILL_PATH = REPO_ROOT / ".github/skills/internal-gateway-execute-plans/SKILL.md" -AGENT_PATH = ( - REPO_ROOT / ".github/skills/internal-gateway-execute-plans/agents/openai.yaml" -) - - -def test_plan_execution_routes_only_preapproved_simplification() -> None: - skill_text = SKILL_PATH.read_text() - - assert "`addyosmani-code-simplification`: plan-bound method owner" in skill_text - assert "current approved plan task explicitly requires" in skill_text - assert "approved review remediation" in skill_text - assert "never introduce it as cleanup outside the approved plan" in skill_text - assert "establish the passing behavior baseline" in skill_text - assert "rerun the same focused validation" in skill_text - - -def test_plan_execution_agent_does_not_preload_simplification() -> None: - assert "addyosmani-code-simplification" not in AGENT_PATH.read_text() diff --git a/tests/github/skills/internal-gateway-execute-plans/test_repository_routing_contract.py b/tests/github/skills/internal-gateway-execute-plans/test_repository_routing_contract.py new file mode 100644 index 00000000..b2c74fcb --- /dev/null +++ b/tests/github/skills/internal-gateway-execute-plans/test_repository_routing_contract.py @@ -0,0 +1,33 @@ +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) + + +def test_execute_plan_handoff_targets_a_real_agent() -> None: + idea = (REPO_ROOT / ".github/agents/internal-gateway-idea.agent.md").read_text() + target = REPO_ROOT / ".github/agents/internal-gateway-execute-plans.agent.md" + assert target.is_file() + assert 'agent: "internal-gateway-execute-plans"' in idea + assert "compact" not in idea.lower() + assert "extended" not in idea.lower() + + +def test_lane_engine_uses_artifact_presence_not_legacy_profiles() -> None: + text = ( + REPO_ROOT / ".github/skills/internal-agent-support-lane-change-engine/SKILL.md" + ).read_text() + assert "approved retained plan" in text + assert "compact" not in text.lower() + assert "extended" not in text.lower() + + +def test_agent_loads_gateway_and_allows_model_invocation() -> None: + text = ( + REPO_ROOT / ".github/agents/internal-gateway-execute-plans.agent.md" + ).read_text() + assert "internal-gateway-execute-plans" in text + assert "disable-model-invocation: true" not in text diff --git a/tests/github/skills/internal-gateway-execute-plans/test_self_containment_contract.py b/tests/github/skills/internal-gateway-execute-plans/test_self_containment_contract.py new file mode 100644 index 00000000..25631a90 --- /dev/null +++ b/tests/github/skills/internal-gateway-execute-plans/test_self_containment_contract.py @@ -0,0 +1,47 @@ +import re +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +BUNDLE = REPO_ROOT / ".github/skills/internal-gateway-execute-plans" +SKILL = BUNDLE / "SKILL.md" +AGENT = BUNDLE / "agents/openai.yaml" + +REQUIRED_CORE_OWNERS = ( + "/superpowers-executing-plans", + "/internal-tdd", + "/superpowers-verification-before-completion", + "/addyosmani-code-simplification", +) +ALLOWED_STATUSES = ("DONE", "PARTIAL", "BLOCKED", "NEEDS_REVIEW") + + +def test_bundle_delegates_execution_and_keeps_local_guardrails() -> None: + skill = SKILL.read_text(encoding="utf-8") + runtime = AGENT.read_text(encoding="utf-8") + combined = f"{skill}\n{runtime}" + assert all(owner in combined for owner in REQUIRED_CORE_OWNERS) + assert "self-contained" not in combined + assert "references/execution-contract.md" in skill + assert "references/status-contract.md" in skill + assert "scripts/plan_execution.py" in skill + + +def test_status_names_and_replacement_scope_are_exact() -> None: + text = (BUNDLE / "references/status-contract.md").read_text() + assert set(re.findall(r"`(DONE|PARTIAL|BLOCKED|NEEDS_REVIEW)`", text)) == set( + ALLOWED_STATUSES + ) + assert ".*.md" not in text + assert "exact allowed sibling filenames" in text + + +def test_runtime_prompt_projects_the_delegated_contract() -> None: + text = AGENT.read_text(encoding="utf-8") + assert all(owner in text for owner in REQUIRED_CORE_OWNERS) + assert "plan fingerprint" in text + assert "fresh task-level evidence" in text + assert "no Git mutation" in text diff --git a/tests/github/skills/internal-gateway-execute-plans/test_validation_entrypoints.py b/tests/github/skills/internal-gateway-execute-plans/test_validation_entrypoints.py new file mode 100644 index 00000000..865edb9d --- /dev/null +++ b/tests/github/skills/internal-gateway-execute-plans/test_validation_entrypoints.py @@ -0,0 +1,62 @@ +import subprocess +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) + + +def test_make_completion_cli_exists() -> None: + makefile = (REPO_ROOT / "Makefile").read_text() + expected = ".github/skills/internal-gateway-execute-plans/scripts/plan_execution.py" + assert expected in makefile + assert (REPO_ROOT / expected).is_file() + + +def test_no_live_legacy_execute_plan_references() -> None: + forbidden = ( + "done-*", + "evidence-envelope.md", + "completion-report.md", + "-plan-state.md", + "Plan profile: compact", + "Plan profile: extended", + ) + paths = ( + REPO_ROOT / ".github/skills/internal-gateway-execute-plans", + REPO_ROOT + / ".github/skills/internal-review-high-level/references/plan-completion-audit.md", + ) + text_parts = [] + for path in paths: + if path.is_file(): + text_parts.append(path.read_text(errors="replace")) + elif path.is_dir(): + for item in path.rglob("*"): + if item.is_file() and item.suffix in {".md", ".yaml", ".py"}: + text_parts.append(item.read_text(errors="replace")) + text = "\n".join(text_parts) + assert not any(marker in text for marker in forbidden) + + +def test_completion_make_target_reaches_bundle_cli(tmp_path: Path) -> None: + plan_file = tmp_path / "plan.md" + plan_file.write_text("# Plan\n") + status_file = tmp_path / "plan.DONE.md" + status_file.write_text("## Status\n`DONE`\n") + result = subprocess.run( + [ + "make", + "retained-plan-check", + f"PLAN_FILE={plan_file}", + f"STATUS_FILE={status_file}", + "PLAN_STAGE=completion", + ], + cwd=REPO_ROOT, + text=True, + capture_output=True, + ) + assert "can't open file" not in result.stderr + assert "No such file or directory" not in result.stderr diff --git a/tests/github/skills/internal-gateway-idea/test_workflow_contract.py b/tests/github/skills/internal-gateway-idea/test_workflow_contract.py index 6ac1835f..871170e3 100644 --- a/tests/github/skills/internal-gateway-idea/test_workflow_contract.py +++ b/tests/github/skills/internal-gateway-idea/test_workflow_contract.py @@ -3,6 +3,8 @@ import sys from pathlib import Path +import yaml + REPO_ROOT = next( parent for parent in Path(__file__).resolve().parents @@ -13,6 +15,7 @@ REPO_ROOT / ".github/skills/internal-gateway-idea/references/workflow.md" ) AGENT_PATH = REPO_ROOT / ".github/skills/internal-gateway-idea/agents/openai.yaml" +ROOT_AGENT_PATH = REPO_ROOT / ".github/agents/internal-gateway-idea.agent.md" SCRIPT_PATH = ( REPO_ROOT / ".github/skills/internal-gateway-idea/scripts/audit_workflow.py" ) @@ -20,6 +23,7 @@ MANDATORY_SEQUENCE = [ "Specialization Checkpoint: gated", "Idea Gate 0", + "External Research Checkpoint", "Assumption Challenge Gate", "Alternative discovery", "Critical Challenge Gate", @@ -43,6 +47,35 @@ def test_bundle_docs_reference_the_scoped_fast_lane() -> None: assert "make internal-gateway-idea-fast-check" in WORKFLOW_PATH.read_text() +def test_runtime_prompt_keeps_the_full_mandatory_gate_sequence() -> None: + _assert_in_order(AGENT_PATH.read_text(), MANDATORY_SEQUENCE) + + +def test_idea_runtime_surfaces_delegate_to_the_expected_owners() -> None: + surfaces = ( + SKILL_PATH.read_text(encoding="utf-8"), + WORKFLOW_PATH.read_text(encoding="utf-8"), + AGENT_PATH.read_text(encoding="utf-8"), + SCRIPT_PATH.read_text(encoding="utf-8"), + ) + for text in surfaces: + assert "/superpowers-brainstorming" in text + assert "/internal-gateway-writing-plans" in text + + +def test_idea_agent_allows_model_invocation() -> None: + frontmatter = yaml.safe_load(ROOT_AGENT_PATH.read_text().split("---", 2)[1]) + assert frontmatter.get("disable-model-invocation") is not True + + +def test_bundle_docs_use_repository_root_validation_commands() -> None: + root_command = ( + "python3 .github/skills/internal-gateway-idea/scripts/audit_workflow.py" + ) + assert root_command in SKILL_PATH.read_text() + assert root_command in WORKFLOW_PATH.read_text() + + def test_audit_workflow_reports_extended_contract_status() -> None: result = subprocess.run( [sys.executable, str(SCRIPT_PATH)], @@ -60,6 +93,62 @@ def test_audit_workflow_reports_extended_contract_status() -> None: assert payload["markers"]["workflow_gate_sequence"] is True assert payload["markers"]["runtime_core_markers"] is True assert payload["markers"]["local_fast_lane_documented"] is True + assert payload["markers"]["compact_chat_projection"] is True + assert payload["markers"]["runtime_gate_sequence"] is True + assert payload["markers"]["runtime_research_checkpoint"] is True + + +def test_idea_bundle_has_compact_user_facing_projection() -> None: + skill_text = SKILL_PATH.read_text() + workflow_text = WORKFLOW_PATH.read_text() + runtime_text = AGENT_PATH.read_text() + + required = [ + "compact user-facing decision card", + "internal workflow state", + "🎯", + "🧭", + "🛠️", + "🧪", + "⚠️", + "✅", + "💡", + "✈️", + ] + for marker in required: + assert marker in skill_text + assert marker in workflow_text + assert marker in runtime_text + + +def test_idea_projection_hides_non_decision_bookkeeping() -> None: + skill_text = SKILL_PATH.read_text() + + assert "Do not announce skipped checkpoints" in skill_text + assert "Do not print the internal gate ledger" in skill_text + assert "four content lines" in skill_text + + +def test_idea_gate_preserves_bulk_questions_and_scopes_compact_cards() -> None: + skill_text = SKILL_PATH.read_text() + workflow_text = WORKFLOW_PATH.read_text() + runtime_text = AGENT_PATH.read_text() + + required = [ + "numbered bulk question block", + "Question", + "Recommendation", + "Why", + "Default if accepted", + "content-bearing output", + ] + for text in (skill_text, workflow_text, runtime_text): + for marker in required: + assert marker in text + + assert "one unresolved decision at a time" not in skill_text + assert "one unresolved decision at a time" not in workflow_text + assert "one unresolved decision at a time" not in runtime_text def test_external_research_checkpoint_is_lazy_and_skill_owned() -> None: @@ -102,12 +191,12 @@ def test_external_research_checkpoint_sits_between_idea_and_challenge() -> None: def test_external_research_checkpoint_has_bounded_outcomes() -> None: - skill_text = SKILL_PATH.read_text() - workflow_text = WORKFLOW_PATH.read_text() + skill_text = SKILL_PATH.read_text(encoding="utf-8") + workflow_text = WORKFLOW_PATH.read_text(encoding="utf-8") outcomes = [ "skip", - "load `mattpocock-research`", + "load `/mattpocock-research`", "one bounded research question", "one Markdown report", "decision-relevant conclusions", @@ -116,3 +205,30 @@ def test_external_research_checkpoint_has_bounded_outcomes() -> None: for marker in outcomes: assert marker in skill_text assert marker in workflow_text + + +def test_spec_is_written_and_reviewed_before_plan_gateway_handoff() -> None: + skill = SKILL_PATH.read_text(encoding="utf-8") + workflow = WORKFLOW_PATH.read_text(encoding="utf-8") + assert "Retained spec writing stays with `/superpowers-brainstorming`" in skill + _assert_in_order( + workflow, + [ + "Write retained spec", + "User reviews retained spec", + "Approve implementation-plan writing", + "Load /internal-gateway-writing-plans", + ], + ) + + +def test_writing_gateway_is_only_an_implementation_plan_owner() -> None: + surfaces = ( + SKILL_PATH.read_text(encoding="utf-8"), + WORKFLOW_PATH.read_text(encoding="utf-8"), + AGENT_PATH.read_text(encoding="utf-8"), + ) + for text in surfaces: + assert "internal-gateway-writing-plans" in text + assert "retained spec or implementation-plan writing" not in text + assert "implementation-plan writing" in surfaces[0] diff --git a/tests/github/skills/internal-gateway-simple-task/test_simple_task_contract.py b/tests/github/skills/internal-gateway-simple-task/test_simple_task_contract.py index fc1c9c32..9019939a 100644 --- a/tests/github/skills/internal-gateway-simple-task/test_simple_task_contract.py +++ b/tests/github/skills/internal-gateway-simple-task/test_simple_task_contract.py @@ -2,6 +2,9 @@ from importlib.util import module_from_spec, spec_from_file_location from pathlib import Path +import pytest +import yaml + REPO_ROOT = next( parent for parent in Path(__file__).resolve().parents @@ -15,6 +18,17 @@ REPO_ROOT / ".github/skills/internal-gateway-simple-task/references/support-routing.md" ) +SIMPLE_LANES_PATH = ( + REPO_ROOT / ".github/skills/internal-gateway-simple-task/references/simple-lanes.md" +) +CLARIFICATION_PATH = ( + REPO_ROOT + / ".github/skills/internal-gateway-simple-task/references/clarification-gate.md" +) +PLAN_MODE_PATH = ( + REPO_ROOT / ".github/skills/internal-gateway-simple-task/references/plan-mode.md" +) +ROOT_AGENT_PATH = REPO_ROOT / ".github/agents/internal-gateway-simple-task.agent.md" RESOLVE_SCRIPT_PATH = ( REPO_ROOT / ".github/skills/internal-gateway-simple-task/scripts/resolve_simple_task.py" @@ -42,10 +56,10 @@ def test_simple_task_bundle_routes_code_changes_to_internal_tdd() -> None: agent_text = AGENT_PATH.read_text() support_routing_text = SUPPORT_ROUTING_PATH.read_text() - assert "`internal-tdd`" in skill_text - assert "load `internal-tdd` before implementation" in skill_text - assert "load `internal-tdd` before implementation" in agent_text - assert "Load `internal-tdd`" in support_routing_text + assert "`/internal-tdd`" in skill_text + assert "load `/internal-tdd` before implementation" in skill_text + assert "load `/internal-tdd` before implementation" in agent_text + assert "Load `/internal-tdd`" in support_routing_text def test_simple_task_skill_has_one_compact_execution_contract() -> None: @@ -113,11 +127,10 @@ def test_trivial_skip_does_not_emit_gate_evidence_ledger() -> None: validation_gap="", ) assert decision["gate_outcome"] == "trivial-skip" - gate_evidence = decision.get("gate_evidence") - assert gate_evidence is None or ( - isinstance(gate_evidence, dict) - and set(gate_evidence.keys()) <= {"validation", "final_evidence"} - ), f"trivial-skip must not emit a 9-row ledger; got {gate_evidence!r}" + assert "gate_evidence" not in decision + gate_requirements = decision.get("gate_requirements") + assert isinstance(gate_requirements, dict) + assert set(gate_requirements.keys()) <= {"validation", "final_evidence"} def test_full_gate_can_require_clarification() -> None: @@ -140,7 +153,7 @@ def test_full_gate_can_require_clarification() -> None: ) assert decision["gate_outcome"] == "full-gate" clarification_row = next( - row for row in decision["gate_evidence"] if row["gate"] == "clarification" + row for row in decision["gate_requirements"] if row["gate"] == "clarification" ) assert clarification_row["required"] is True @@ -199,7 +212,8 @@ def test_suggest_does_not_emit_worktree_mapping() -> None: def test_code_simplification_requires_explicit_authorization() -> None: skill_text = SKILL_PATH.read_text() - assert "`addyosmani-code-simplification`: on-demand method owner" in skill_text + assert "`/addyosmani-code-simplification`: on-demand method owner" in skill_text + assert "load `/addyosmani-code-simplification`" in skill_text assert "explicit code-simplification request" in skill_text assert "already-approved simplification remediation" in skill_text assert "establish a passing behavior baseline" in skill_text @@ -209,3 +223,447 @@ def test_code_simplification_requires_explicit_authorization() -> None: ) assert "Only the five referenced skill names appear in this bundle." in skill_text assert "addyosmani-code-simplification" not in AGENT_PATH.read_text() + + +def _full_gate_decision() -> dict[str, object]: + return resolve_simple_task.build_gate_decision( + task="Enable the global Copilot link", + lane="edit", + trivial_kind=None, + prompt="", + depth_keywords=[], + risks=[], + needs_plan=False, + needs_review=False, + needs_critical=False, + owner_ambiguous=False, + clarification_overflow=False, + validation_obvious=False, + validation_path="pytest -q tests/example.py", + validation_gap="", + ) + + +def test_default_gate_text_is_compact(capsys) -> None: + resolve_simple_task.render_gate_text(_full_gate_decision()) + + lines = capsys.readouterr().out.splitlines() + assert lines == [ + "🧭 full-gate: Enable the global Copilot link", + "🛠️ Scope: Single-lane edit work.", + "🧪 Check: pytest -q tests/example.py", + ( + "⚠️ Risk: Task still fits one bounded run but needs the full gate" + " before action." + ), + ] + + +def test_trivial_gate_text_needs_no_extra_approval(capsys) -> None: + decision = resolve_simple_task.build_gate_decision( + task="Fix a typo", + lane="edit", + trivial_kind="tiny-edit", + prompt="", + depth_keywords=[], + risks=[], + needs_plan=False, + needs_review=False, + needs_critical=False, + owner_ambiguous=False, + clarification_overflow=False, + validation_obvious=False, + validation_path="make docs-lint", + validation_gap="", + ) + + resolve_simple_task.render_gate_text(decision) + + output = capsys.readouterr().out + assert len(output.splitlines()) == 4 + assert "✈️ Action:" not in output + assert "Readiness Brief:" not in output + assert "Gate Evidence:" not in output + + +def test_stopped_gate_text_surfaces_blocker_and_action(capsys) -> None: + decision = resolve_simple_task.build_gate_decision( + task="Redesign the workflow", + lane="edit", + trivial_kind=None, + prompt="", + depth_keywords=[], + risks=[], + needs_plan=True, + needs_review=False, + needs_critical=False, + owner_ambiguous=False, + clarification_overflow=False, + validation_obvious=False, + validation_path="make skill-lint", + validation_gap="", + ) + + resolve_simple_task.render_gate_text(decision) + + lines = capsys.readouterr().out.splitlines() + assert len(lines) == 4 + assert lines[0].startswith("🧭 Stop:") + assert "plan-recommended" in lines[1] + assert "✈️ Action:" in lines[3] + + +def test_gate_cli_json_retains_complete_internal_requirements() -> None: + import json + import subprocess + + result = subprocess.run( + [ + sys.executable, + str(RESOLVE_SCRIPT_PATH), + "gate", + "--task", + "Enable the global Copilot link", + "--lane", + "edit", + "--validation-path", + "pytest -q tests/example.py", + "--format", + "json", + ], + cwd=REPO_ROOT, + text=True, + capture_output=True, + check=False, + ) + + assert result.returncode == 0, result.stdout + result.stderr + payload = json.loads(result.stdout) + assert payload["readiness_brief"]["anti_scope"] + assert payload["readiness_brief"]["stop_conditions"] + assert payload["gate_requirements"] + assert "gate_evidence" not in payload + + +def test_gate_cli_approval_required_stops() -> None: + import json + import subprocess + + result = subprocess.run( + [ + sys.executable, + str(RESOLVE_SCRIPT_PATH), + "gate", + "--task", + "Execute an approval-bound rollout", + "--lane", + "edit", + "--approval-required", + "--validation-path", + "Run the approved rollout check", + "--format", + "json", + ], + cwd=REPO_ROOT, + text=True, + capture_output=True, + check=False, + ) + + assert result.returncode == 0, result.stdout + result.stderr + payload = json.loads(result.stdout) + assert payload["gate_outcome"] == "stop-with-reason" + assert "approval-required" in payload["reason_codes"] + + +def test_simple_task_bundle_documents_compact_projection() -> None: + skill_text = SKILL_PATH.read_text() + runtime_text = AGENT_PATH.read_text() + + for marker in [ + "compact user-facing projection", + "internal readiness record", + "🎯", + "🧭", + "🛠️", + "🧪", + "⚠️", + "✅", + "💡", + "✈️", + ]: + assert marker in skill_text + + for marker in [ + "gate requirements", + "no more than four content lines", + "approval boundary", + ]: + assert marker in runtime_text + + assert "normal chat must not dump" in skill_text + assert "`--format json`" in skill_text + + +def test_nontrivial_validation_gap_stops_before_execution() -> None: + decision = resolve_simple_task.build_gate_decision( + task="Update an unvalidated contract", + lane="edit", + trivial_kind=None, + prompt="", + depth_keywords=[], + risks=[], + needs_plan=False, + needs_review=False, + needs_critical=False, + owner_ambiguous=False, + clarification_overflow=False, + validation_obvious=False, + validation_path="", + validation_gap="No local validator", + ) + + assert decision["gate_outcome"] == "stop-with-reason" + assert decision["next_action"] == "stop" + assert "validation-gap" in decision["reason_codes"] + + +def test_security_risk_with_validation_gap_stops_before_execution() -> None: + decision = resolve_simple_task.build_gate_decision( + task="Change a security-sensitive policy", + lane="edit", + trivial_kind=None, + prompt="", + depth_keywords=[], + risks=["security"], + needs_plan=False, + needs_review=False, + needs_critical=False, + owner_ambiguous=False, + clarification_overflow=False, + validation_obvious=False, + validation_path="", + validation_gap="No local security check", + ) + + assert decision["gate_outcome"] == "stop-with-reason" + assert decision["next_action"] == "stop" + assert "material-risk:security" in decision["reason_codes"] + assert "validation-gap" in decision["reason_codes"] + + +def test_unnamed_obvious_validation_cannot_enable_trivial_skip() -> None: + decision = resolve_simple_task.build_gate_decision( + task="Tiny edit", + lane="edit", + trivial_kind="tiny-edit", + prompt="", + depth_keywords=[], + risks=[], + needs_plan=False, + needs_review=False, + needs_critical=False, + owner_ambiguous=False, + clarification_overflow=False, + validation_obvious=True, + validation_path="", + validation_gap="", + ) + + assert decision["gate_outcome"] != "trivial-skip" + assert decision["next_action"] == "stop" + assert "validation-path-missing" in decision["reason_codes"] + + +def test_full_gate_does_not_pause_for_unrequested_approval() -> None: + decision = _full_gate_decision() + + assert decision["gate_outcome"] == "full-gate" + assert decision["next_action"] == "execute" + assert decision["needs_explicit_approval"] is False + + +def test_gate_decision_names_requirements_without_claiming_evidence() -> None: + decision = _full_gate_decision() + + assert "gate_evidence" not in decision + requirements = decision["gate_requirements"] + assert {row["gate"] for row in requirements} == set(resolve_simple_task.GATE_ROWS) + assert all("expected_evidence" in row for row in requirements) + assert all("status" not in row for row in requirements) + + +def test_full_gate_text_has_at_most_four_lines_and_no_extra_approval(capsys) -> None: + resolve_simple_task.render_gate_text(_full_gate_decision()) + + lines = capsys.readouterr().out.splitlines() + assert len(lines) <= 4 + assert not any("Confirm before" in line for line in lines) + assert any("pytest -q tests/example.py" in line for line in lines) + + +def test_stop_text_names_boundary_evidence_and_action(capsys) -> None: + decision = resolve_simple_task.build_gate_decision( + task="Broad plan", + lane="edit", + trivial_kind=None, + prompt="", + depth_keywords=[], + risks=[], + needs_plan=True, + needs_review=False, + needs_critical=False, + owner_ambiguous=False, + clarification_overflow=False, + validation_obvious=False, + validation_path="make skill-lint", + validation_gap="", + ) + + resolve_simple_task.render_gate_text(decision) + output = capsys.readouterr().out + lines = output.splitlines() + + assert len(lines) <= 4 + assert "retained plan" in output + assert "plan-recommended" in output + assert "Provide" in output + assert "✈️" in output + assert "still fits one bounded run" not in output + + +def test_skill_exposes_each_conditional_reference_with_a_context_pointer() -> None: + skill_text = SKILL_PATH.read_text() + + for reference in [ + "references/clarification-gate.md", + "references/plan-mode.md", + "references/simple-lanes.md", + "references/support-routing.md", + ]: + assert reference in skill_text + + +def test_runtime_prompt_matches_clarification_and_approval_contract() -> None: + skill_text = SKILL_PATH.read_text() + runtime_text = AGENT_PATH.read_text() + + required = [ + "only when a missing bounded fact blocks the active lane", + "approval boundary", + "gate requirements", + "no more than four content lines", + ] + for marker in required: + assert marker in skill_text + assert marker in runtime_text + + assert "Use `grill-me` when the task is non-trivial" not in runtime_text + assert "Always surface" not in runtime_text + + +def test_lane_reference_keeps_gate_evidence_internal() -> None: + lanes_text = SIMPLE_LANES_PATH.read_text() + + assert "`gate-ledger`" not in lanes_text + assert "compact user-facing projection" in lanes_text + + +def test_clarification_reference_has_no_broken_stop_rules_pointer() -> None: + clarification_text = CLARIFICATION_PATH.read_text() + + assert "Stop rules above" not in clarification_text + assert "Stop Conditions above" in clarification_text + + +@pytest.mark.parametrize( + ("overrides", "expected_outcome"), + [ + ( + { + "task": "Answer from one local file", + "lane": "answer", + "trivial_kind": "focused-read", + "validation_path": "Cite the inspected file", + }, + "trivial-skip", + ), + ( + { + "task": "Fix one reproduced unit-test failure", + "lane": "diagnose", + "validation_path": "pytest -q tests/example.py", + }, + "full-gate", + ), + ( + { + "task": "Review a pull request for findings", + "lane": "validate", + "needs_review": True, + "validation_path": "Inspect the existing diff", + }, + "stop-with-reason", + ), + ( + { + "task": "Design a cross-cutting governance workflow", + "lane": "edit", + "risks": ["architecture", "governance"], + "validation_gap": "Design direction is not approved", + }, + "stop-with-reason", + ), + ( + { + "task": "Execute an approval-bound rollout", + "lane": "edit", + "approval_required": True, + "validation_path": "Run the approved rollout check", + }, + "stop-with-reason", + ), + ], +) +def test_gate_boundary_matrix( + overrides: dict[str, object], + expected_outcome: str, +) -> None: + defaults: dict[str, object] = { + "task": "bounded task", + "lane": "edit", + "trivial_kind": None, + "prompt": "", + "depth_keywords": [], + "risks": [], + "needs_plan": False, + "needs_review": False, + "needs_critical": False, + "owner_ambiguous": False, + "clarification_overflow": False, + "needs_clarification": False, + "validation_obvious": False, + "validation_path": "", + "validation_gap": "", + "approval_required": False, + } + defaults.update(overrides) + + decision = resolve_simple_task.build_gate_decision(**defaults) + assert decision["gate_outcome"] == expected_outcome + + +def _frontmatter(path: Path) -> dict[str, object]: + raw_text = path.read_text() + _, yaml_text, _ = raw_text.split("---", 2) + parsed = yaml.safe_load(yaml_text) + assert isinstance(parsed, dict) + return parsed + + +def test_skill_and_agent_allow_model_invocation() -> None: + skill_frontmatter = _frontmatter(SKILL_PATH) + agent_frontmatter = _frontmatter(ROOT_AGENT_PATH) + + assert skill_frontmatter.get("disable-model-invocation") is not True + assert agent_frontmatter.get("disable-model-invocation") is not True diff --git a/tests/github/skills/internal-gateway-writing-plans/test_validate_plan.py b/tests/github/skills/internal-gateway-writing-plans/test_validate_plan.py new file mode 100644 index 00000000..6069232b --- /dev/null +++ b/tests/github/skills/internal-gateway-writing-plans/test_validate_plan.py @@ -0,0 +1,65 @@ +import importlib.util +import shutil +import subprocess +import sys +import tempfile +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +BUNDLE = REPO_ROOT / ".github/skills/internal-gateway-writing-plans" +SCRIPT = BUNDLE / "scripts/validate_plan.py" +VALID = BUNDLE / "fixtures/2026-07-25-1829-valid-plan.md" +INVALID = BUNDLE / "fixtures/2026-07-25-1829-invalid-plan.md" + + +def _module(): + spec = importlib.util.spec_from_file_location("validate_plan", SCRIPT) + assert spec and spec.loader + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_valid_fixture_has_no_objective_findings() -> None: + assert _module().validate_plan(VALID) == [] + + +def test_invalid_fixture_reports_every_objective_rule() -> None: + with tempfile.TemporaryDirectory() as tmpdir: + invalid_copy = Path(tmpdir) / "invalid-name.md" + shutil.copy(INVALID, invalid_copy) + codes = {finding["code"] for finding in _module().validate_plan(invalid_copy)} + assert codes == { + "filename", + "preflight", + "ordered_tasks", + "file_targets", + "validation", + "git_mutation", + "execution_owner", + } + + +def test_cli_is_quiet_on_success_and_bounded_on_failure() -> None: + passed = subprocess.run( + [sys.executable, str(SCRIPT), str(VALID)], + cwd=REPO_ROOT, + text=True, + capture_output=True, + check=False, + ) + failed = subprocess.run( + [sys.executable, str(SCRIPT), str(INVALID)], + cwd=REPO_ROOT, + text=True, + capture_output=True, + check=False, + ) + assert passed.returncode == 0 + assert "PASS" in passed.stdout + assert failed.returncode == 1 + assert failed.stdout.count("\n") <= 10 diff --git a/tests/github/skills/internal-gateway-writing-plans/test_writing_plans_contract.py b/tests/github/skills/internal-gateway-writing-plans/test_writing_plans_contract.py index 09ac60dc..b2a13980 100644 --- a/tests/github/skills/internal-gateway-writing-plans/test_writing_plans_contract.py +++ b/tests/github/skills/internal-gateway-writing-plans/test_writing_plans_contract.py @@ -1,5 +1,7 @@ from pathlib import Path +import yaml + REPO_ROOT = next( parent for parent in Path(__file__).resolve().parents @@ -11,15 +13,104 @@ ) -def test_skill_requires_hhmm_filenames_for_retained_artifacts() -> None: - text = SKILL_PATH.read_text() +def _skill_frontmatter() -> dict[str, object]: + text = SKILL_PATH.read_text(encoding="utf-8") + return yaml.safe_load(text.split("---", 2)[1]) - assert "tmp/superpowers/plans/YYYY-MM-DD-HHMM-.md" in text - assert "tmp/superpowers/specs/YYYY-MM-DD-HHMM--design.md" in text +def _agent_prompt() -> str: + payload = yaml.safe_load(AGENT_PATH.read_text(encoding="utf-8")) + return payload["interface"]["default_prompt"] -def test_agent_prompt_mentions_hhmm_filenames_for_plan_and_spec() -> None: - text = AGENT_PATH.read_text() +def test_skill_requires_hhmm_filenames_for_retained_plans() -> None: + text = SKILL_PATH.read_text(encoding="utf-8") + assert "tmp/superpowers/plans/YYYY-MM-DD-HHMM-.md" in text + assert "tmp/superpowers/specs/" not in text + + +def test_agent_prompt_mentions_hhmm_filenames_for_plans_only() -> None: + text = AGENT_PATH.read_text(encoding="utf-8") assert "YYYY-MM-DD-HHMM-.md" in text - assert "YYYY-MM-DD-HHMM--design.md" in text + assert "specs/" not in text + + +def test_description_targets_only_approved_implementation_planning() -> None: + description = str(_skill_frontmatter()["description"]) + assert "approved implementation plan" in description + assert "retained writing" not in description + assert "spec writing" not in description + + +def test_spec_authoring_is_owned_upstream() -> None: + skill = SKILL_PATH.read_text(encoding="utf-8") + assert "Retained-spec writing stays in the brainstorming lane" in skill + assert "Use after the user approves implementation-plan writing" in skill + assert "specs use" not in skill + + +def test_runtime_prompt_carries_the_narrowed_boundary() -> None: + prompt = _agent_prompt() + required = ( + "approved implementation plan", + "delegated draft", + "local acceptance gate", + "internal-gateway-execute-plans", + "No-Commit Rule", + ) + for marker in required: + assert marker in prompt + assert "specs/" not in prompt + + +def test_writing_runtime_surfaces_delegate_to_expected_owners() -> None: + for path in (SKILL_PATH, AGENT_PATH): + text = path.read_text(encoding="utf-8") + assert "/superpowers-writing-plans" in text + assert "/internal-gateway-execute-plans" in text + + +GATES = ( + "Preflight Gate", + "Delegated Draft Gate", + "Local Acceptance Gate", + "Writing Stop", +) + + +def _assert_in_order(text: str, markers: tuple[str, ...]) -> None: + positions = [text.index(marker) for marker in markers] + assert positions == sorted(positions) + + +def test_skill_uses_ordered_gates_with_completion_criteria() -> None: + skill = SKILL_PATH.read_text(encoding="utf-8") + _assert_in_order(skill, GATES) + assert skill.count("Completion criterion:") == len(GATES) + + +def test_delegated_output_is_draft_until_local_acceptance() -> None: + skill = SKILL_PATH.read_text(encoding="utf-8") + assert "Delegated output remains draft-only" in skill + assert "objective checks pass" in skill + assert "human judgment checks pass" in skill + assert "revise the draft in place" in skill + + +def test_local_acceptance_enforces_authoring_discipline() -> None: + skill = SKILL_PATH.read_text(encoding="utf-8") + required = ( + "coherent task boundaries", + "one responsibility per task", + "focused validation", + "Reject unapproved simplification or duplicated execution workflow.", + ) + for marker in required: + assert marker in skill + assert "## Execution Discipline" not in skill + + +def test_accepted_plan_routes_to_the_repository_execution_gateway() -> None: + skill = SKILL_PATH.read_text(encoding="utf-8") + assert "`/internal-gateway-execute-plans`" in skill + assert "Stop after reporting the accepted plan path" in skill diff --git a/tests/github/skills/internal-gcp/test_internal_gcp_routing_contract.py b/tests/github/skills/internal-gcp/test_internal_gcp_routing_contract.py new file mode 100644 index 00000000..f9a27b08 --- /dev/null +++ b/tests/github/skills/internal-gcp/test_internal_gcp_routing_contract.py @@ -0,0 +1,196 @@ +import re +from pathlib import Path + +import yaml + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SKILL_DIR = REPO_ROOT / ".github/skills/internal-gcp" +LEGACY_SKILL_DIR = REPO_ROOT / ".github/skills" / ("internal-gcp-" + "strategic") +SKILL_PATH = SKILL_DIR / "SKILL.md" +AGENT_PATH = SKILL_DIR / "agents/openai.yaml" +ROUTING_MATRIX_PATH = SKILL_DIR / "references/routing-matrix.md" +LENS_PLAYBOOK_PATH = SKILL_DIR / "references/lens-playbook.md" + +EXPECTED_DESCRIPTION = ( + "Use when a Google Cloud task cannot be routed confidently to a specific GCP " + "skill because the request is materially ambiguous, has multiple GCP domains " + "with no clear primary owner, or requires clarification before selecting the " + "correct specialist, or when the user needs high-level Google Cloud platform " + "decision support or tradeoff framing before implementation. Do not use for " + "clearly scoped organization structure, governance or IAM, or operations or " + "validation tasks." +) + + +def load_frontmatter(path: Path) -> dict[str, object]: + _, raw_frontmatter, _ = path.read_text().split("---", maxsplit=2) + return yaml.safe_load(raw_frontmatter) + + +def test_internal_gcp_replaces_the_legacy_bundle() -> None: + assert SKILL_PATH.is_file() + assert not LEGACY_SKILL_DIR.exists() + assert load_frontmatter(SKILL_PATH) == { + "name": "internal-gcp", + "description": EXPECTED_DESCRIPTION, + } + + +def test_internal_gcp_contract_is_router_and_strategic() -> None: + skill_text = SKILL_PATH.read_text() + + router_markers = ( + "material routing uncertainty", + "Do not activate only because the task concerns Google Cloud", + "Do not activate when one specialist clearly owns the next step", + "Select the minimum specialist set", + "Explicit `$internal-gcp` invocation remains valid", + ) + for marker in router_markers: + assert marker in skill_text + + strategic_markers = ( + "Identify the decision first, not the implementation tool", + "Compare realistic options, not strawmen.", + "Keep tradeoffs concrete.", + ) + for marker in strategic_markers: + assert marker in skill_text + + +def test_internal_gcp_interface_names_router_and_strategic() -> None: + interface = yaml.safe_load(AGENT_PATH.read_text())["interface"] + + assert interface == { + "display_name": "Internal GCP", + "short_description": "GCP routing and strategic decision support", + "default_prompt": ( + "Use $internal-gcp to route an unclear GCP task to the minimum " + "specialist set, or to frame a Google Cloud decision when the next " + "step is not yet structure, governance, operations, or delivery." + ), + } + + +def test_routing_matrix_covers_positive_negative_and_multi_domain_cases() -> None: + matrix_text = ROUTING_MATRIX_PATH.read_text() + + for heading in ( + "## Fallback-positive cases", + "## Direct-specialist negative cases", + "## Multi-domain primary-owner cases", + "## Review rule", + ): + assert heading in matrix_text + + +def test_lens_playbook_keeps_strategic_depth() -> None: + playbook_text = LENS_PLAYBOOK_PATH.read_text() + + for heading in ( + "## Common lens combinations", + "## Decision note pattern", + "## Depth control", + ): + assert heading in playbook_text + + +GCP_SKILL_PATHS = sorted((REPO_ROOT / ".github/skills").glob("internal-gcp*/SKILL.md")) +LEGACY_SKILL_ID = "internal-gcp-" + "strategic" +FORBIDDEN_GENERIC_REFERENCES = ( + "internal-bash-script", + "internal-python-script", + "internal-python", + "internal-python-project", + "internal-nodejs", + "internal-nodejs-project", + "internal-terraform", +) + +EXPECTED_SPECIALIST_DESCRIPTION_PREFIXES = { + "internal-gcp-organization-structure": "Use when ", + "internal-gcp-governance": "Use when ", + "internal-gcp-operations": "Use when ", +} + + +def test_gcp_family_has_no_legacy_or_generic_skill_references() -> None: + assert len(GCP_SKILL_PATHS) == 4 + + for path in GCP_SKILL_PATHS: + skill_text = path.read_text() + assert LEGACY_SKILL_ID not in skill_text + for forbidden_name in FORBIDDEN_GENERIC_REFERENCES: + assert f"`{forbidden_name}`" not in skill_text + + +def test_specialists_name_internal_gcp_only_as_uncertainty_fallback() -> None: + specialist_paths = [path for path in GCP_SKILL_PATHS if path != SKILL_PATH] + + for path in specialist_paths: + skill_text = path.read_text() + assert "`internal-gcp`" in skill_text + assert "material routing uncertainty" in skill_text + + +LANE_SKILL_IDS = ( + "internal-gcp-governance", + "internal-gcp-operations", + "internal-gcp-organization-structure", +) + +SKILL_REFERENCE_PATTERN = re.compile( + r"`((?:internal|awesome|openai|superpowers|agent-os|antigravity|addyosmani" + r"|local|mattpocock|terraform|vercel|customize|grill|graphify)-[a-z0-9-]+)`" +) + +REMOVED_SECTION_HEADINGS = ( + "## Handoffs", + "## Cross-references", + "## Referenced skills", + "## Relationship to adjacent skills", + "## When not to use", +) + + +def test_lane_skills_have_no_sibling_references_or_handoffs() -> None: + for skill_id in LANE_SKILL_IDS: + skill_dir = REPO_ROOT / ".github/skills" / skill_id + text_paths = [ + skill_dir / "SKILL.md", + *sorted(skill_dir.glob("references/*.md")), + ] + for text_path in text_paths: + skill_text = text_path.read_text() + for heading in REMOVED_SECTION_HEADINGS: + assert heading not in skill_text, f"{skill_id} keeps {heading}" + references = set(SKILL_REFERENCE_PATTERN.findall(skill_text)) + assert references <= {"internal-gcp"}, ( + f"{text_path.name} references {sorted(references)}" + ) + assert "handoff" not in skill_text.lower(), ( + f"{text_path.name} still mentions handoffs" + ) + + +def test_specialist_descriptions_carry_positive_and_negative_triggers() -> None: + for path in GCP_SKILL_PATHS: + frontmatter = load_frontmatter(path) + name = frontmatter["name"] + if name == "internal-gcp": + continue + assert name in EXPECTED_SPECIALIST_DESCRIPTION_PREFIXES + description = frontmatter["description"] + assert description.startswith(EXPECTED_SPECIALIST_DESCRIPTION_PREFIXES[name]) + assert "Do not use" in description + + +def test_inventory_lists_only_the_canonical_internal_gcp_bundle() -> None: + inventory_text = (REPO_ROOT / ".github/INVENTORY.md").read_text() + + assert ".github/skills/internal-gcp/SKILL.md" in inventory_text + assert LEGACY_SKILL_ID not in inventory_text diff --git a/tests/github/skills/internal-github/test_internal_github_routing_contract.py b/tests/github/skills/internal-github/test_internal_github_routing_contract.py new file mode 100644 index 00000000..51901d84 --- /dev/null +++ b/tests/github/skills/internal-github/test_internal_github_routing_contract.py @@ -0,0 +1,182 @@ +import re +from pathlib import Path + +import yaml + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SKILL_DIR = REPO_ROOT / ".github/skills/internal-github" +LEGACY_SKILL_DIR = REPO_ROOT / ".github/skills" / ("internal-github-" + "strategic") +SKILL_PATH = SKILL_DIR / "SKILL.md" +AGENT_PATH = SKILL_DIR / "agents/openai.yaml" +ROUTING_MATRIX_PATH = SKILL_DIR / "references/routing-matrix.md" +STRATEGIC_FRAMING_PATH = SKILL_DIR / "references/strategic-framing.md" + +EXPECTED_DESCRIPTION = ( + "Use when a GitHub task cannot be routed confidently to a specific GitHub " + "skill because the request is materially ambiguous, has multiple GitHub " + "domains with no clear primary owner, or requires clarification before " + "selecting the correct specialist, or when the user needs high-level GitHub " + "platform or operating-model decision support or tradeoff framing before " + "implementation. Do not use for clearly scoped governance, operations, PR " + "lifecycle, Actions workflow authoring, composite-action authoring, or " + "current Copilot platform behavior research." +) + + +def load_frontmatter(path: Path) -> dict[str, object]: + _, raw_frontmatter, _ = path.read_text().split("---", maxsplit=2) + return yaml.safe_load(raw_frontmatter) + + +def test_internal_github_replaces_the_legacy_bundle() -> None: + assert SKILL_PATH.is_file() + assert not LEGACY_SKILL_DIR.exists() + assert load_frontmatter(SKILL_PATH) == { + "name": "internal-github", + "description": EXPECTED_DESCRIPTION, + } + + +def test_internal_github_contract_is_router_and_strategic() -> None: + skill_text = SKILL_PATH.read_text() + + router_markers = ( + "material routing uncertainty", + "Do not activate only because the task concerns GitHub", + "Do not activate when one specialist clearly owns the next step", + "Select the minimum specialist set", + "Explicit `$internal-github` invocation remains valid", + ) + for marker in router_markers: + assert marker in skill_text + + strategic_markers = ( + "Identify the decision first, not the implementation tool", + "Compare realistic options, not strawmen.", + "Keep tradeoffs concrete.", + ) + for marker in strategic_markers: + assert marker in skill_text + + +def test_internal_github_interface_names_router_and_strategic() -> None: + interface = yaml.safe_load(AGENT_PATH.read_text())["interface"] + + assert interface == { + "display_name": "Internal GitHub", + "short_description": "GitHub routing and strategic decision support", + "default_prompt": ( + "Use $internal-github to route an unclear GitHub task to the " + "minimum specialist set, or to frame a GitHub platform or " + "operating-model decision when the next step is not yet " + "governance, operations, or delivery." + ), + } + + +def test_routing_matrix_covers_positive_negative_and_multi_domain_cases() -> None: + matrix_text = ROUTING_MATRIX_PATH.read_text() + + for heading in ( + "## Fallback-positive cases", + "## Direct-specialist negative cases", + "## Multi-domain primary-owner cases", + "## Review rule", + ): + assert heading in matrix_text + + +def test_strategic_framing_keeps_strategic_depth() -> None: + framing_text = STRATEGIC_FRAMING_PATH.read_text() + + for heading in ( + "## Common lens combinations", + "## Decision note pattern", + "## Depth control", + ): + assert heading in framing_text + + +GITHUB_SKILL_PATHS = sorted( + (REPO_ROOT / ".github/skills").glob("internal-github*/SKILL.md") +) +LEGACY_SKILL_ID = "internal-github-" + "strategic" +FORBIDDEN_GENERIC_REFERENCES = ( + "internal-bash-script", + "internal-python-script", + "internal-python", + "internal-python-project", + "internal-nodejs", + "internal-nodejs-project", + "internal-terraform", +) + + +def test_github_family_has_no_legacy_or_generic_skill_references() -> None: + assert len(GITHUB_SKILL_PATHS) == 6 + + for path in GITHUB_SKILL_PATHS: + skill_text = path.read_text() + assert LEGACY_SKILL_ID not in skill_text + for forbidden_name in FORBIDDEN_GENERIC_REFERENCES: + assert f"`{forbidden_name}`" not in skill_text + + +def test_specialists_name_internal_github_only_as_uncertainty_fallback() -> None: + specialist_paths = [path for path in GITHUB_SKILL_PATHS if path != SKILL_PATH] + + for path in specialist_paths: + skill_text = path.read_text() + assert "`internal-github`" in skill_text + assert "material routing uncertainty" in skill_text + + +LANE_SKILL_IDS = ( + "internal-github-governance", + "internal-github-operations", + "internal-github-pr", +) + +SKILL_REFERENCE_PATTERN = re.compile( + r"`((?:internal|awesome|openai|superpowers|agent-os|antigravity|addyosmani" + r"|local|mattpocock|terraform|vercel|customize|grill|graphify)-[a-z0-9-]+)`" +) + +REMOVED_SECTION_HEADINGS = ( + "## Handoffs", + "## Cross-references", + "## Referenced skills", + "## Relationship to adjacent skills", + "## When not to use", +) + + +def test_lane_skills_have_no_sibling_references_or_handoffs() -> None: + for skill_id in LANE_SKILL_IDS: + skill_dir = REPO_ROOT / ".github/skills" / skill_id + text_paths = [ + skill_dir / "SKILL.md", + *sorted(skill_dir.glob("references/*.md")), + ] + for text_path in text_paths: + skill_text = text_path.read_text() + for heading in REMOVED_SECTION_HEADINGS: + assert heading not in skill_text, f"{skill_id} keeps {heading}" + references = set(SKILL_REFERENCE_PATTERN.findall(skill_text)) + assert references <= {"internal-github"}, ( + f"{text_path.name} references {sorted(references)}" + ) + assert "handoff" not in skill_text.lower(), ( + f"{text_path.name} still mentions handoffs" + ) + + +def test_inventory_lists_only_the_canonical_internal_github_bundle() -> None: + inventory_text = (REPO_ROOT / ".github/INVENTORY.md").read_text() + + assert ".github/skills/internal-github/SKILL.md" in inventory_text + assert LEGACY_SKILL_ID not in inventory_text diff --git a/tests/github/skills/internal-skill-creator/test_cross_skill_invocation_contract.py b/tests/github/skills/internal-skill-creator/test_cross_skill_invocation_contract.py new file mode 100644 index 00000000..04e5db14 --- /dev/null +++ b/tests/github/skills/internal-skill-creator/test_cross_skill_invocation_contract.py @@ -0,0 +1,84 @@ +import re +from pathlib import Path + +import yaml + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SKILLS_ROOT = REPO_ROOT / ".github" / "skills" +CREATOR_ROOT = SKILLS_ROOT / "internal-skill-creator" + +CALLED_SKILLS = { + "internal-skill-creator": { + "mattpocock-writing-great-skills", + }, + "internal-gateway-codebase-improvement": { + "addyosmani-code-simplification", + "internal-tdd", + "mattpocock-improve-codebase-architecture", + "superpowers-verification-before-completion", + }, + "internal-gateway-idea": { + "internal-gateway-writing-plans", + "mattpocock-research", + "superpowers-brainstorming", + }, + "internal-gateway-simple-task": { + "addyosmani-code-simplification", + "grill-me", + "internal-gateway-critical-master", + "internal-tdd", + "superpowers-verification-before-completion", + }, + "internal-gateway-writing-plans": { + "internal-gateway-execute-plans", + "superpowers-writing-plans", + }, + "internal-gateway-execute-plans": { + "addyosmani-code-simplification", + "internal-tdd", + "superpowers-executing-plans", + "superpowers-verification-before-completion", + }, +} + + +def _instruction_text(skill_name: str) -> str: + root = SKILLS_ROOT / skill_name + paths = [root / "SKILL.md", root / "agents" / "openai.yaml"] + paths.extend(sorted((root / "references").glob("*.md"))) + return "\n".join( + path.read_text(encoding="utf-8") for path in paths if path.exists() + ) + + +def _frontmatter(skill_name: str) -> dict[str, object]: + text = (SKILLS_ROOT / skill_name / "SKILL.md").read_text(encoding="utf-8") + return yaml.safe_load(text.split("---", 2)[1]) + + +def test_skill_creator_defines_slash_prefixed_cross_skill_invocation() -> None: + text = (CREATOR_ROOT / "SKILL.md").read_text(encoding="utf-8") + assert "Prefix every cross-skill invocation with `/`" in text + + +def test_creator_and_gateway_calls_are_slash_prefixed() -> None: + for caller, callees in CALLED_SKILLS.items(): + text = _instruction_text(caller) + for callee in callees: + unprefixed = re.compile(rf"(? None: + for callee in set().union(*CALLED_SKILLS.values()): + frontmatter = _frontmatter(callee) + assert frontmatter.get("disable-model-invocation") is not True, ( + f"{callee} is called by another skill but blocks model invocation" + ) diff --git a/tests/github/skills/internal-skill-creator/test_internal_skill_creator_contract.py b/tests/github/skills/internal-skill-creator/test_internal_skill_creator_contract.py index ae9cea60..81be5bb3 100644 --- a/tests/github/skills/internal-skill-creator/test_internal_skill_creator_contract.py +++ b/tests/github/skills/internal-skill-creator/test_internal_skill_creator_contract.py @@ -5,97 +5,100 @@ for parent in Path(__file__).resolve().parents if (parent / "AGENTS.md").exists() and (parent / ".github").exists() ) -SKILL_PATH = REPO_ROOT / ".github/skills/internal-skill-creator/SKILL.md" -CHECKLIST_PATH = ( - REPO_ROOT - / ".github/skills/internal-skill-creator/references/writing-skills-checklist.md" -) - - -def test_referenced_skills_are_audit_index_not_preload() -> None: - skill_text = SKILL_PATH.read_text() - checklist_text = CHECKLIST_PATH.read_text() - - assert "audit index, not a preload" in skill_text - assert "audit index, not a preload" in checklist_text - assert "Do not load referenced skills from this section alone" in checklist_text - - -def test_generic_skill_shape_is_conditional_not_rigid() -> None: - checklist_text = CHECKLIST_PATH.read_text() +BUNDLE_ROOT = REPO_ROOT / ".github/skills/internal-skill-creator" +SKILL_PATH = BUNDLE_ROOT / "SKILL.md" +OPENAI_PATH = BUNDLE_ROOT / "agents/openai.yaml" +REFERENCE_PATH = BUNDLE_ROOT / "references" / "authoring-and-evaluation.md" - assert "## Generic skill shape" in checklist_text - assert "Conditional sections" in checklist_text - assert "Do not require every section for every skill" in checklist_text +def workflow_text() -> str: + return SKILL_PATH.read_text(encoding="utf-8").split("## Workflow", 1)[1] -def test_skill_cleanup_preserves_triggers_and_removes_responsibility_duplication() -> ( - None -): - checklist_text = CHECKLIST_PATH.read_text() - assert ( - "Remove duplicated responsibility, not useful trigger reinforcement" - in checklist_text +def test_material_work_uses_core_method_before_local_evaluation() -> None: + workflow = workflow_text() + headings = ( + "### 1. Repository preflight", + "### 2. Core authoring and revision", + "### 3. Proportional evaluation", + "### 4. Repository closure", ) - assert "Preserve a working `description:` during cleanup" in checklist_text - - -def test_skill_md_does_not_restate_checklist() -> None: - import re - - skill_text = SKILL_PATH.read_text() - checklist_text = CHECKLIST_PATH.read_text() - - def normalize(text: str) -> str: - return re.sub(r"\s+", " ", text).strip().lower() - - skill_norm = normalize(skill_text) - checklist_norm = normalize(checklist_text) - - shared_phrases = [ - "iron law: do not create or materially revise a skill without first seeing the failure", - "treat skills as reusable reference guides, not narratives", - "prefer the smallest change that fixes the local problem", - "keep `description:` trigger-only", - "preserve a working `description:` during token optimization", - "treat generic skill shape as conditional, not a rigid section template", - "treat `## referenced skills` as an audit index, not a preload list", - "remove duplicated responsibility, not useful trigger reinforcement", - "prefer `references/` over new `scripts/` for static tables", - "reference other skills by skill name and behavior only", - "prefer bundle-relative references to files under", - "do not copy the same material back into `skill.md`", - "compare the wrapper against its core before editing", - ] - - duplicates = [ - phrase - for phrase in shared_phrases - if phrase in skill_norm and phrase in checklist_norm - ] - - assert not duplicates, ( - f"SKILL.md restates {len(duplicates)} phrases also in checklist: {duplicates[:3]}" + positions = [workflow.index(heading) for heading in headings] + assert positions == sorted(positions) + assert "Load `/mattpocock-writing-great-skills`" in workflow + assert "core method" in workflow + + +def test_local_evaluation_stage_is_evidence_gated() -> None: + workflow = workflow_text() + evaluation = workflow.split("### 3. Proportional evaluation", 1)[1].split( + "### 4. Repository closure", 1 + )[0] + normalized = " ".join(evaluation.split()) + assert "references/authoring-and-evaluation.md" in normalized + assert "applicable evaluation branches" in normalized + assert "skipped branches and reasons" in normalized + assert "evidence, blockers, and completion status" in normalized + + +def test_local_authoring_reference_covers_the_retained_contract() -> None: + reference = REFERENCE_PATH.read_text(encoding="utf-8") + required_markers = ( + "## Intent contract", + "## Evaluation selection", + "## Baselines", + "## Evidence and human review", + "## Description trigger checks", + "## Iteration stop conditions", ) - - -def test_core_backed_wrapper_guidance_is_generic_and_reference_owned() -> None: - skill_text = SKILL_PATH.read_text() - checklist_text = CHECKLIST_PATH.read_text() - - required_guidance = ( - "## Core-backed wrappers", - "Compare the wrapper against its core before editing", - "trigger, repository-local policy, and proven environment fallbacks", - "Do not restate the core's workflow, decision logic, output contract, or validation procedure", - "Structural validation is not semantic alignment", - "paired agent", + for marker in required_markers: + assert marker in reference + assert "objective" in reference.lower() + assert "subjective" in reference.lower() + assert "near-miss" in reference.lower() + assert "holdout" in reference.lower() + + +def test_internal_bundle_has_only_the_core_skill_dependency() -> None: + bundle_text = "\n".join( + path.read_text(encoding="utf-8") + for path in BUNDLE_ROOT.rglob("*") + if path.is_file() ) - - for phrase in required_guidance: - assert phrase in checklist_text - - assert "Compare the wrapper against its core before editing" not in skill_text - assert "internal-review-code" not in checklist_text - assert "addyosmani-code-review-and-quality" not in checklist_text + assert "anthropic-skill-creator" not in bundle_text + assert "local-agent-sync-external-resources" not in bundle_text + assert "internal-agent-creator" not in bundle_text + assert bundle_text.count("mattpocock-writing-great-skills") >= 2 + + +def test_core_stage_revises_instead_of_only_reporting() -> None: + workflow = workflow_text() + review = workflow.split("### 2. Core authoring and revision", 1)[1].split( + "### 3. Proportional evaluation", 1 + )[0] + assert "revise the draft" in review + assert "invocation, description, information hierarchy" in review + assert "duplication, sediment, and no-ops" in review + + +def test_local_closure_keeps_repository_specific_checks() -> None: + workflow = workflow_text() + closure = workflow.split("### 4. Repository closure", 1)[1] + assert "agents/openai.yaml" in closure + assert "validate_internal_skills" in closure + assert "routing fallout" in closure + assert "before/after" in closure + + +def test_redundant_local_references_are_removed() -> None: + assert not (BUNDLE_ROOT / "references/writing-skills-checklist.md").exists() + assert not (BUNDLE_ROOT / "references/script-output-contract.md").exists() + + +def test_default_prompt_names_core_method_before_local_closure() -> None: + prompt = OPENAI_PATH.read_text(encoding="utf-8") + matt = prompt.index("mattpocock-writing-great-skills") + evaluation = prompt.index("proportional evaluation") + closure = prompt.index("repository closure") + assert matt < evaluation < closure + assert "anthropic-skill-creator" not in prompt diff --git a/tests/github/skills/local-agent-sync-external-resources/scripts/test_candidate.py b/tests/github/skills/local-agent-sync-external-resources/scripts/test_candidate.py index 4a1eee5c..d5c05c9b 100644 --- a/tests/github/skills/local-agent-sync-external-resources/scripts/test_candidate.py +++ b/tests/github/skills/local-agent-sync-external-resources/scripts/test_candidate.py @@ -62,6 +62,18 @@ def git_repo(tmp_path: Path) -> Path: return repo +def _write_source_metadata(sources_root: Path, source: ManagedSource) -> None: + source_dir = sources_root / source.source_id + source_dir.mkdir(parents=True, exist_ok=True) + upstream_paths = sorted(asset.upstream for asset in source.assets) + digest = hashlib.sha256(",".join(upstream_paths).encode("utf-8")).hexdigest() + tsv = ( + f"source_id\trepository\tref\tpaths_sha256\n" + f"{source.source_id}\t{source.repository}\t{source.ref}\t{digest}\n" + ) + (source_dir / ".external-resource-source.tsv").write_text(tsv, encoding="utf-8") + + def _example_asset(local: str = ".github/skills/example") -> ManagedAsset: return ManagedAsset( source="test", @@ -82,6 +94,7 @@ def _superpowers_resources() -> ManagedResources: source_id="obra-superpowers", repository="https://github.com/obra/superpowers.git", ref="abc123", + advertised_ref=None, assets=(asset,), ) replacement = TextReplacement( @@ -96,6 +109,46 @@ def _superpowers_resources() -> ManagedResources: ) +def _mattpocock_resources() -> ManagedResources: + assets = ( + ManagedAsset( + source="mattpocock-skills", + upstream="skills/engineering/tdd", + local=".github/skills/mattpocock-tdd", + canonical_name="mattpocock-tdd", + ), + ManagedAsset( + source="mattpocock-skills", + upstream="skills/engineering/grill-with-docs", + local=".github/skills/mattpocock-grill-with-docs", + canonical_name="mattpocock-grill-with-docs", + ), + ManagedAsset( + source="mattpocock-skills", + upstream="skills/engineering/domain-modeling", + local=".github/skills/mattpocock-domain-modeling", + canonical_name="mattpocock-domain-modeling", + ), + ) + source = ManagedSource( + source_id="mattpocock-skills", + repository="https://github.com/mattpocock/skills.git", + ref="abc123", + advertised_ref=None, + assets=assets, + rewrite_skill_references=True, + backtick_skill_references=("tdd",), + ) + replacement = TextReplacement( + source="mattpocock-skills", + old="/grilling", + new="/grill-me", + ) + return ManagedResources( + sources=(source,), replacements=(replacement,), watchlist=() + ) + + def test_workspace_inside_repository_is_rejected(tmp_path: Path) -> None: repo = tmp_path / "repo" workspace = repo / "tmp" / "refresh" @@ -159,11 +212,89 @@ def test_normalization_updates_name_and_declared_text_only( assert "superpowers-verification-before-completion" in content +@pytest.mark.parametrize( + "canonical_name", + ("superpowers-brainstorming", "grill-me"), +) +def test_normalization_enforces_guided_bulk_questions_for_interview_skills( + tmp_path: Path, + canonical_name: str, +) -> None: + candidate = tmp_path / "candidate" + local = f".github/skills/{canonical_name}" + skill = candidate / local / "SKILL.md" + skill.parent.mkdir(parents=True) + skill.write_text( + f"---\nname: {canonical_name}\n---\nAsk clarifying questions one at a time.\n", + encoding="utf-8", + ) + asset = ManagedAsset( + source="upstream", + upstream=f"skills/{canonical_name}", + local=local, + canonical_name=canonical_name, + ) + resources = ManagedResources( + sources=( + ManagedSource( + source_id="upstream", + repository="https://example.com/upstream.git", + ref="a" * 40, + advertised_ref=None, + assets=(asset,), + ), + ), + replacements=(), + watchlist=(), + ) + + first_changed = normalize_candidate(resources, candidate) + second_changed = normalize_candidate(resources, candidate) + + content = skill.read_text(encoding="utf-8") + assert first_changed == (f"{local}/SKILL.md",) + assert second_changed == () + assert content.count("Local guided-question contract") == 1 + assert "numbered bulk question blocks" in content + assert "`Question`, `Recommendation`, `Why`, and `Default if accepted`" in content + assert "Keep each question, recommendation, and reason brief" in content + assert "overrides any earlier instruction to ask one question at a time" in content + + +def test_normalization_rewrites_declared_mattpocock_skill_references( + tmp_path: Path, +) -> None: + candidate = tmp_path / "candidate" + skill = candidate / ".github/skills/mattpocock-tdd/SKILL.md" + skill.parent.mkdir(parents=True) + skill.write_text( + "---\nname: tdd\n---\n" + "Use /domain-modeling, /tdd, /grill-with-docs, and /grilling.\n" + "See `tdd` and /unmanaged when needed.\n", + encoding="utf-8", + ) + + changed = normalize_candidate(_mattpocock_resources(), candidate) + + assert changed == (".github/skills/mattpocock-tdd/SKILL.md",) + content = skill.read_text(encoding="utf-8") + assert "name: mattpocock-tdd" in content + assert "/mattpocock-tdd" in content + assert "`mattpocock-tdd`" in content + assert "/mattpocock-grill-with-docs" in content + assert "/grill-me" in content + assert "/mattpocock-domain-modeling" in content + assert "/grilling" not in content + assert "/domain-modeling" not in content + assert "/unmanaged" in content + + def test_materialize_candidate_copies_upstream_to_local( tmp_path: Path, ) -> None: workspace = tmp_path / "workspace" - source_checkout = workspace / "sources" / "test-source" / "skills" / "example" + sources_root = workspace / "sources" + source_checkout = sources_root / "test-source" / "skills" / "example" source_checkout.mkdir(parents=True) (source_checkout / "SKILL.md").write_text( "---\nname: upstream-name\n---\n", encoding="utf-8" @@ -178,9 +309,11 @@ def test_materialize_candidate_copies_upstream_to_local( source = ManagedSource( source_id="test-source", repository="https://example.com/repo.git", - ref="abc", + ref="a" * 40, + advertised_ref=None, assets=(asset,), ) + _write_source_metadata(sources_root, source) resources = ManagedResources(sources=(source,), replacements=(), watchlist=()) candidate = tmp_path / "candidate" @@ -348,6 +481,7 @@ def test_build_candidate_patch_detects_repo_vs_candidate_diff( source_id="test-source", repository="https://example.com/repo.git", ref="abc", + advertised_ref=None, assets=(asset,), ), ), @@ -381,9 +515,11 @@ def test_materialize_candidate_uses_explicit_source_root( source = ManagedSource( source_id="test-source", repository="https://example.com/repo.git", - ref="abc", + ref="a" * 40, + advertised_ref=None, assets=(asset,), ) + _write_source_metadata(external_sources, source) resources = ManagedResources(sources=(source,), replacements=(), watchlist=()) candidate = tmp_path / "candidate" @@ -416,21 +552,24 @@ def test_materialize_candidate_reports_all_missing_upstreams( local=".github/skills/b", canonical_name="b", ) + source_a = ManagedSource( + source_id="src-a", + repository="https://example.com/a.git", + ref="a" * 40, + advertised_ref=None, + assets=(asset_a,), + ) + source_b = ManagedSource( + source_id="src-b", + repository="https://example.com/b.git", + ref="b" * 40, + advertised_ref=None, + assets=(asset_b,), + ) + _write_source_metadata(sources_root, source_a) + _write_source_metadata(sources_root, source_b) resources = ManagedResources( - sources=( - ManagedSource( - source_id="src-a", - repository="https://example.com/a.git", - ref="abc", - assets=(asset_a,), - ), - ManagedSource( - source_id="src-b", - repository="https://example.com/b.git", - ref="def", - assets=(asset_b,), - ), - ), + sources=(source_a, source_b), replacements=(), watchlist=(), ) @@ -458,12 +597,8 @@ def test_materialize_candidate_reports_expected_source_root(tmp_path: Path) -> N materialize_candidate(resources, workspace, candidate) message = str(excinfo.value) - assert "Missing upstream paths:" in message - assert (workspace / "sources").as_posix() in message - assert ( - "Prepare the missing source checkout under that root or pass --source-root." - in message - ) + assert "Missing prepared source metadata:" in message + assert "Run prepare before audit/plan/apply." in message def test_load_overrides_rejects_missing_patch_file(tmp_path: Path) -> None: @@ -523,3 +658,125 @@ def test_override_3way_replay_uses_real_git_repo( assert len(results) == 1 assert results[0].status == "applied" assert "Patched." in target.read_text(encoding="utf-8") + + +def test_grill_with_docs_normalizes_to_mattpocock_wrapper_delegating_to_grill_me( + tmp_path: Path, +) -> None: + workspace = tmp_path / "workspace" + sources_root = workspace / "sources" + grill_dir = ( + sources_root + / "mattpocock-skills" + / "skills" + / "engineering" + / "grill-with-docs" + ) + grill_dir.mkdir(parents=True) + (grill_dir / "SKILL.md").write_text( + "---\nname: grill-with-docs\n---\n" + "Run a `/grilling` session, using the `/domain-modeling` skill.\n", + encoding="utf-8", + ) + domain_dir = ( + sources_root + / "mattpocock-skills" + / "skills" + / "engineering" + / "domain-modeling" + ) + domain_dir.mkdir(parents=True) + (domain_dir / "SKILL.md").write_text( + "---\nname: domain-modeling\n---\nDomain modeling.\n", + encoding="utf-8", + ) + + assets = ( + ManagedAsset( + source="mattpocock-skills", + upstream="skills/engineering/grill-with-docs", + local=".github/skills/mattpocock-grill-with-docs", + canonical_name="mattpocock-grill-with-docs", + ), + ManagedAsset( + source="mattpocock-skills", + upstream="skills/engineering/domain-modeling", + local=".github/skills/mattpocock-domain-modeling", + canonical_name="mattpocock-domain-modeling", + ), + ) + source = ManagedSource( + source_id="mattpocock-skills", + repository="https://github.com/mattpocock/skills.git", + ref="abc123", + advertised_ref=None, + assets=assets, + rewrite_skill_references=True, + ) + replacement = TextReplacement( + source="mattpocock-skills", + old="/grilling", + new="/grill-me", + ) + resources = ManagedResources( + sources=(source,), replacements=(replacement,), watchlist=() + ) + _write_source_metadata(sources_root, source) + + candidate = tmp_path / "candidate" + materialize_candidate(resources, workspace, candidate) + normalize_candidate(resources, candidate) + + wrapper = candidate / ".github/skills/mattpocock-grill-with-docs/SKILL.md" + assert wrapper.exists() + content = wrapper.read_text(encoding="utf-8") + assert "name: mattpocock-grill-with-docs" in content + assert "/grill-me" in content + assert "/mattpocock-domain-modeling" in content + assert "/grilling" not in content + assert "/mattpocock-grill-with-docs session" not in content + + +def test_undeclared_backtick_reference_is_left_unchanged(tmp_path: Path) -> None: + candidate = tmp_path / "candidate" + skill = candidate / ".github/skills/mattpocock-domain-modeling/SKILL.md" + skill.parent.mkdir(parents=True) + skill.write_text( + "---\nname: domain-modeling\n---\n" + "Use `domain-modeling` for vocabulary and `tdd` for the loop.\n" + "See /domain-modeling for the command form.\n", + encoding="utf-8", + ) + + normalize_candidate(_mattpocock_resources(), candidate) + + content = skill.read_text(encoding="utf-8") + assert "`domain-modeling`" in content + assert "`mattpocock-domain-modeling`" not in content + assert "`mattpocock-tdd`" in content + assert "/mattpocock-domain-modeling" in content + + +def _example_asset() -> ManagedAsset: + return ManagedAsset( + source="test-source", + upstream="skills/example", + local=".github/skills/example", + canonical_name="example", + ) + + +def test_renamed_managed_file_is_reported_once_with_new_path(git_repo: Path) -> None: + target = git_repo / ".github/skills/example" + target.mkdir(parents=True) + (target / "SKILL.md").write_text("---\nname: example\n---\n", encoding="utf-8") + _commit_all(git_repo) + + _run_git( + git_repo, + ["mv", ".github/skills/example/SKILL.md", ".github/skills/example/RENAMED.md"], + ) + + dirty = find_dirty_targets(git_repo, (_example_asset(),)) + + assert dirty == (".github/skills/example/RENAMED.md",) diff --git a/tests/github/skills/local-agent-sync-external-resources/scripts/test_cli.py b/tests/github/skills/local-agent-sync-external-resources/scripts/test_cli.py index 34d85f85..3e6e3eb2 100644 --- a/tests/github/skills/local-agent-sync-external-resources/scripts/test_cli.py +++ b/tests/github/skills/local-agent-sync-external-resources/scripts/test_cli.py @@ -14,6 +14,21 @@ sys.path.insert(0, SCRIPT_DIR.as_posix()) +def _write_source_metadata_for_fixture( + sources_root: Path, source_id: str, repository: str, ref: str, upstream: str +) -> None: + source_dir = sources_root / source_id + source_dir.mkdir(parents=True, exist_ok=True) + import hashlib + + digest = hashlib.sha256(upstream.encode("utf-8")).hexdigest() + tsv = ( + f"source_id\trepository\tref\tpaths_sha256\n" + f"{source_id}\t{repository}\t{ref}\t{digest}\n" + ) + (source_dir / ".external-resource-source.tsv").write_text(tsv, encoding="utf-8") + + def _run_git(cwd: Path, args: list[str]) -> None: subprocess.run( ["git", *args], @@ -54,7 +69,7 @@ def test_audit_does_not_fetch_or_write(repo_root: Path) -> None: payload = json.loads(result.stdout) assert payload["mode"] == "audit" assert payload["repository_changed"] is False - assert payload["managed_assets"] == 45 + assert payload["managed_assets"] == 56 def test_apply_refuses_dirty_target(tmp_path: Path) -> None: @@ -71,7 +86,7 @@ def test_apply_refuses_dirty_target(tmp_path: Path) -> None: sources: test-source: repository: https://example.com/repo.git - ref: abc123 + ref: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa assets: - upstream: skills/example local: .github/skills/example @@ -137,7 +152,7 @@ def test_apply_reports_repository_changed_when_candidate_diff_applies( sources: test-source: repository: https://example.com/repo.git - ref: abc123 + ref: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa assets: - upstream: skills/example local: .github/skills/example @@ -168,6 +183,13 @@ def test_apply_reports_repository_changed_when_candidate_diff_applies( (source_dir / "SKILL.md").write_text( "---\nname: example\n---\nNew content.\n", encoding="utf-8" ) + _write_source_metadata_for_fixture( + workspace / "sources", + "test-source", + "https://example.com/repo.git", + "a" * 40, + "skills/example", + ) result = subprocess.run( [ @@ -210,7 +232,7 @@ def test_plan_uses_explicit_source_root(tmp_path: Path) -> None: sources: test-source: repository: https://example.com/repo.git - ref: abc123 + ref: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa assets: - upstream: skills/example local: .github/skills/example @@ -236,13 +258,19 @@ def test_plan_uses_explicit_source_root(tmp_path: Path) -> None: workspace = tmp_path / "external-workspace" workspace.mkdir() - external_sources = ( - tmp_path / "external-sources" / "test-source" / "skills" / "example" - ) + external_sources_root = tmp_path / "external-sources" + external_sources = external_sources_root / "test-source" / "skills" / "example" external_sources.mkdir(parents=True) (external_sources / "SKILL.md").write_text( "---\nname: example\n---\nNew content.\n", encoding="utf-8" ) + _write_source_metadata_for_fixture( + external_sources_root, + "test-source", + "https://example.com/repo.git", + "a" * 40, + "skills/example", + ) result = subprocess.run( [ @@ -254,7 +282,7 @@ def test_plan_uses_explicit_source_root(tmp_path: Path) -> None: "--workspace", str(workspace), "--source-root", - str(tmp_path / "external-sources"), + str(external_sources_root), "--manifest", str(manifest_src), "--overrides", @@ -286,7 +314,7 @@ def test_plan_then_apply_end_to_end(tmp_path: Path) -> None: sources: test-source: repository: https://example.com/repo.git - ref: abc123 + ref: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa assets: - upstream: skills/example local: .github/skills/example @@ -317,6 +345,13 @@ def test_plan_then_apply_end_to_end(tmp_path: Path) -> None: (source_dir / "SKILL.md").write_text( "---\nname: example\n---\nNew content.\n", encoding="utf-8" ) + _write_source_metadata_for_fixture( + workspace / "sources", + "test-source", + "https://example.com/repo.git", + "a" * 40, + "skills/example", + ) common_args = [ sys.executable, @@ -355,6 +390,155 @@ def test_plan_then_apply_end_to_end(tmp_path: Path) -> None: assert "New content." in target.read_text(encoding="utf-8") +def test_audit_tsv_contains_summary_mode_row(repo_root: Path) -> None: + result = subprocess.run( + [ + sys.executable, + str(SCRIPT_DIR / "sync_external_resources.py"), + "audit", + "--repo-root", + str(repo_root), + "--format", + "tsv", + ], + capture_output=True, + text=True, + check=False, + ) + + assert result.returncode == 0 + assert "summary\tmode\tok\taudit" in result.stdout + + +def test_prepare_cold_then_warm_against_fixture( + tmp_path: Path, +) -> None: + repo = tmp_path / "repo" + repo.mkdir() + _run_git(repo, ["init"]) + _run_git(repo, ["config", "user.email", "test@test.com"]) + _run_git(repo, ["config", "user.name", "Test"]) + _commit_all(repo) + + remote = tmp_path / "remote.git" + remote.mkdir() + _run_git(remote, ["init", "--bare"]) + _run_git(remote, ["config", "uploadpack.allowReachableSHA1InWant", "true"]) + _run_git(remote, ["config", "uploadpack.allowFilter", "true"]) + + work = tmp_path / "work" + work.mkdir() + _run_git(work, ["init"]) + _run_git(work, ["config", "user.email", "test@test.com"]) + _run_git(work, ["config", "user.name", "Test"]) + + skill_dir = work / "skills" / "example" + skill_dir.mkdir(parents=True) + (skill_dir / "SKILL.md").write_text( + "---\nname: example\n---\nHello.\n", encoding="utf-8" + ) + + decoy = work / "decoy.bin" + decoy.write_bytes(b"\x00" * 1024) + + _commit_all(work) + sha_result = subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=work, + capture_output=True, + text=True, + check=True, + ) + commit_sha = sha_result.stdout.strip() + _run_git(work, ["remote", "add", "origin", str(remote)]) + _run_git(work, ["push", "origin", "HEAD:refs/heads/main"]) + + manifest_src = tmp_path / "manifest.yaml" + manifest_src.write_text( + f"""\ +version: 1 +sources: + test-source: + repository: {remote} + ref: {commit_sha} + assets: + - upstream: skills/example + local: .github/skills/example + canonical_name: example +watchlist: [] +""", + encoding="utf-8", + ) + overrides_src = tmp_path / "overrides.yaml" + overrides_src.write_text("version: 1\noverrides: []\n", encoding="utf-8") + + workspace = tmp_path / "external-workspace" + workspace.mkdir() + + common_args = [ + sys.executable, + str(SCRIPT_DIR / "sync_external_resources.py"), + "--repo-root", + str(repo), + "--workspace", + str(workspace), + "--manifest", + str(manifest_src), + "--overrides", + str(overrides_src), + "--format", + "tsv", + ] + + first = subprocess.run( + [*common_args, "prepare"], + capture_output=True, + text=True, + check=False, + ) + assert first.returncode == 0, first.stderr + assert "source\ttest-source" in first.stdout + assert "metric\ttest-source.materialized_files\tok" in first.stdout + + snapshot_skill = ( + workspace / "sources" / "test-source" / "skills" / "example" / "SKILL.md" + ) + assert snapshot_skill.exists() + assert not (workspace / "sources" / "test-source" / "decoy.bin").exists() + + second = subprocess.run( + [*common_args, "prepare"], + capture_output=True, + text=True, + check=False, + ) + assert second.returncode == 0, second.stderr + assert "source\ttest-source\tcached" in second.stdout + + +def test_owner_docs_state_prepare_is_the_only_network_mode( + repo_root: Path, +) -> None: + skill_md = repo_root / ".github/skills/local-agent-sync-external-resources/SKILL.md" + agent_md = repo_root / ".github/agents/local-sync-external-resources.agent.md" + texts = [] + for path in (skill_md, agent_md): + if path.exists(): + texts.append(path.read_text(encoding="utf-8")) + combined = "\n".join(texts) + + if not combined: + pytest.skip("Owner docs not yet written") + + for phrase in ( + "prepare", + "offline", + "pinned", + "no package", + ): + assert phrase.lower() in combined.lower(), f"Owner docs must mention {phrase!r}" + + def test_bundle_exposes_one_public_cli(repo_root: Path) -> None: scripts = repo_root / ".github/skills/local-agent-sync-external-resources/scripts" public_scripts = sorted( @@ -378,7 +562,7 @@ def test_audit_reports_dirty_targets_but_stays_zero_exit(tmp_path: Path) -> None sources: test-source: repository: https://example.com/repo.git - ref: abc123 + ref: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa assets: - upstream: skills/example local: .github/skills/example @@ -437,7 +621,7 @@ def test_plan_missing_sources_names_explicit_source_root(tmp_path: Path) -> None sources: test-source: repository: https://example.com/repo.git - ref: abc123 + ref: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa assets: - upstream: skills/example local: .github/skills/example @@ -492,7 +676,6 @@ def test_agent_and_skill_do_not_route_to_unneeded_skills(repo_root: Path) -> Non pytest.skip("Agent or skill file not yet rewritten") for forbidden in ( - "openai-skill-creator", "internal-skill-creator", "internal-agent-creator", "internal-gateway-idea", @@ -500,3 +683,76 @@ def test_agent_and_skill_do_not_route_to_unneeded_skills(repo_root: Path) -> Non "internal-copilot-audit", ): assert forbidden not in text + + +def test_invalid_manifest_emits_blocker_not_traceback( + tmp_path: Path, repo_root: Path +) -> None: + bad_manifest = tmp_path / "bad-manifest.yaml" + bad_manifest.write_text("version: 2\nsources: {}\n", encoding="utf-8") + + result = subprocess.run( + [ + sys.executable, + str(SCRIPT_DIR / "sync_external_resources.py"), + "audit", + "--repo-root", + str(repo_root), + "--manifest", + str(bad_manifest), + "--format", + "tsv", + ], + capture_output=True, + text=True, + check=False, + ) + + assert result.returncode == 2 + assert "Traceback" not in result.stderr + assert result.stdout.splitlines()[0] == "record\tkey\tstatus\tvalue" + assert "blocker\t" in result.stdout + assert "version 1" in result.stdout + + +def test_prepare_tsv_metric_rows_use_status_column_for_status( + tmp_path: Path, repo_root: Path +) -> None: + from source_prepare_core import PrepareSourceResult # noqa: E402 + from sync_external_resources import SyncOutcome # noqa: E402 + + outcome = SyncOutcome( + mode="prepare", + workspace="/tmp/ws", + managed_assets=1, + changed_paths=(), + override_results=(), + validations=(), + blockers=(), + repository_changed=False, + source_results=( + PrepareSourceResult( + source_id="example", + repository="https://example.com/repo.git", + ref="a" * 40, + cache_status="fetched", + fetch_strategy="direct-sha", + materialized_files=3, + materialized_bytes=42, + cache_bytes_added=7, + duration_ms=5, + ), + ), + ) + + rows = { + (record.record, record.key): (record.status, record.value) + for record in outcome.to_records() + } + + assert rows[("metric", "example.materialized_files")] == ("ok", "3") + assert rows[("metric", "example.materialized_bytes")] == ("ok", "42") + assert rows[("metric", "example.cache_bytes_added")] == ("ok", "7") + assert rows[("metric", "example.duration_ms")] == ("ok", "5") + assert rows[("validation", "example.fetch_strategy")] == ("ok", "direct-sha") + assert rows[("source", "example")] == ("fetched", "a" * 40) diff --git a/tests/github/skills/local-agent-sync-external-resources/scripts/test_manifest.py b/tests/github/skills/local-agent-sync-external-resources/scripts/test_manifest.py index 68118128..426c0913 100644 --- a/tests/github/skills/local-agent-sync-external-resources/scripts/test_manifest.py +++ b/tests/github/skills/local-agent-sync-external-resources/scripts/test_manifest.py @@ -1,3 +1,5 @@ +import hashlib +import re import sys from pathlib import Path @@ -17,20 +19,199 @@ validate_override_patches, ) +_COMMIT_OBJECT_ID_RE = re.compile(r"^(?:[0-9a-f]{40}|[0-9a-f]{64})$") + +_FULL_SHA40 = "a" * 40 +_FULL_SHA40_ALT = "b" * 40 + @pytest.fixture def repo_root() -> Path: return REPO_ROOT +def _write_manifest(tmp_path: Path, body: str) -> Path: + path = tmp_path / "managed-resources.yaml" + path.write_text(body, encoding="utf-8") + return path + + +def test_live_manifest_refs_are_full_lowercase_object_ids(repo_root: Path) -> None: + manifest = load_managed_resources( + repo_root + / ".github/skills/local-agent-sync-external-resources/references/managed-resources.yaml" + ) + for source in manifest.sources: + assert _COMMIT_OBJECT_ID_RE.match(source.ref), ( + f"source {source.source_id} ref {source.ref!r} " + f"is not a full lowercase commit object ID" + ) + + +def test_manifest_accepts_optional_advertised_ref(tmp_path: Path) -> None: + manifest = load_managed_resources( + _write_manifest( + tmp_path, + f"""\ +version: 1 +sources: + source: + repository: https://github.com/example/repo.git + ref: {_FULL_SHA40} + advertised_ref: refs/heads/main + assets: + - upstream: skills/one + local: .github/skills/example + canonical_name: example +watchlist: [] +""", + ) + ) + assert manifest.sources[0].advertised_ref == "refs/heads/main" + + +def test_manifest_rejects_short_ref(tmp_path: Path) -> None: + with pytest.raises(ValueError, match="full lowercase commit object ID"): + load_managed_resources( + _write_manifest( + tmp_path, + """\ +version: 1 +sources: + source: + repository: https://github.com/example/repo.git + ref: abc123 + assets: + - upstream: skills/one + local: .github/skills/example + canonical_name: example +watchlist: [] +""", + ) + ) + + +def test_manifest_rejects_branch_name_ref(tmp_path: Path) -> None: + with pytest.raises(ValueError, match="full lowercase commit object ID"): + load_managed_resources( + _write_manifest( + tmp_path, + """\ +version: 1 +sources: + source: + repository: https://github.com/example/repo.git + ref: main + assets: + - upstream: skills/one + local: .github/skills/example + canonical_name: example +watchlist: [] +""", + ) + ) + + +def test_manifest_rejects_uppercase_sha_ref(tmp_path: Path) -> None: + with pytest.raises(ValueError, match="full lowercase commit object ID"): + load_managed_resources( + _write_manifest( + tmp_path, + f"""\ +version: 1 +sources: + source: + repository: https://github.com/example/repo.git + ref: {"A" * 40} + assets: + - upstream: skills/one + local: .github/skills/example + canonical_name: example +watchlist: [] +""", + ) + ) + + def test_live_manifest_preserves_declared_scope(repo_root: Path) -> None: manifest = load_managed_resources( repo_root / ".github/skills/local-agent-sync-external-resources/references/managed-resources.yaml" ) - assert len(manifest.assets) == 45 + assert len(manifest.assets) == 56 assert len(manifest.watchlist) == 13 + matt_source = next( + source for source in manifest.sources if source.source_id == "mattpocock-skills" + ) + assert matt_source.rewrite_skill_references is True + assert dict(matt_source.skill_reference_aliases) == {} + assert { + item.upstream_id + for item in manifest.watchlist + if item.source_family == "mattpocock/skills" + } >= {"prototype", "triage", "to-tickets", "qa"} + assert {item.canonical_name for item in matt_source.assets} >= { + "mattpocock-grill-with-docs", + "mattpocock-domain-modeling", + "mattpocock-codebase-design", + "mattpocock-improve-codebase-architecture", + "mattpocock-implement", + "mattpocock-tdd", + "mattpocock-to-spec", + "mattpocock-setup-matt-pocock-skills", + "mattpocock-code-review", + "mattpocock-wayfinder", + "mattpocock-writing-great-skills", + } + assert "grill-me" not in {item.canonical_name for item in matt_source.assets} + assert { + (source.repository, asset.upstream, asset.local, asset.canonical_name) + for source in manifest.sources + for asset in source.assets + } >= { + ( + "https://github.com/atlassian/atlassian-mcp-server.git", + "skills/search-company-knowledge", + ".github/skills/search-company-knowledge", + "search-company-knowledge", + ), + ( + "https://github.com/openai/skills.git", + "skills/.curated/openai-docs", + ".github/skills/openai-docs", + "openai-docs", + ), + ( + "https://github.com/anthropics/skills.git", + "skills/docx", + ".github/skills/anthropic-docx", + "anthropic-docx", + ), + ( + "https://github.com/anthropics/skills.git", + "skills/pptx", + ".github/skills/anthropic-pptx", + "anthropic-pptx", + ), + ( + "https://github.com/anthropics/skills.git", + "skills/xlsx", + ".github/skills/anthropic-xlsx", + "anthropic-xlsx", + ), + } + imported_assets = { + (source.repository, asset.upstream, asset.local, asset.canonical_name) + for source in manifest.sources + for asset in source.assets + } + assert ( + "https://github.com/anthropics/skills.git", + "skills/skill-creator", + ".github/skills/anthropic-skill-creator", + "anthropic-skill-creator", + ) not in imported_assets assert { item.local for item in manifest.assets if item.source == "obra-superpowers" } == { @@ -61,12 +242,12 @@ def test_live_manifest_preserves_declared_scope(repo_root: Path) -> None: def test_manifest_rejects_duplicate_local_paths(tmp_path: Path) -> None: path = tmp_path / "managed-resources.yaml" path.write_text( - """\ + f"""\ version: 1 sources: source: repository: https://github.com/example/repo.git - ref: abc123 + ref: {_FULL_SHA40} assets: - upstream: skills/one local: .github/skills/example @@ -136,3 +317,51 @@ def test_live_handoff_override_forces_repo_tmp_handoff(repo_root: Path) -> None: assert handoff.override_id == "mattpocock-handoff-tmp-path" patch_text = (bundle_root / handoff.patch_path).read_text(encoding="utf-8") assert "tmp/handoff/" in patch_text + + +def test_mattpocock_skill_creator_review_keeps_invocation_override( + repo_root: Path, +) -> None: + overrides_path = ( + repo_root / ".github/skills/local-agent-sync-external-resources/references/" + "imported-asset-overrides.yaml" + ) + bundle_root = repo_root / ".github/skills/local-agent-sync-external-resources" + overrides = load_overrides(overrides_path) + by_target = {override.target_path: override for override in overrides} + + expected = { + ".github/skills/mattpocock-writing-great-skills/SKILL.md": "mattpocock-writing-great-skills-delegated-invocation", + } + for target, override_id in expected.items(): + override = by_target[target] + assert override.override_id == override_id + digest = hashlib.sha256((repo_root / target).read_bytes()).hexdigest() + assert override.expected_content_hash == digest + patch = (bundle_root / override.patch_path).read_text(encoding="utf-8") + assert "-disable-model-invocation: true" in patch + assert "internal-skill-creator" in patch + assert ".github/skills/anthropic-skill-creator/SKILL.md" not in by_target + + +def test_manifest_rejects_undeclared_backtick_skill_reference(tmp_path: Path) -> None: + manifest = tmp_path / "managed-resources.yaml" + manifest.write_text( + "version: 1\n" + "sources:\n" + " example-source:\n" + " repository: https://example.com/repo.git\n" + f" ref: {'a' * 40}\n" + " rewrite_skill_references: true\n" + " backtick_skill_references:\n" + " - not-an-asset\n" + " assets:\n" + " - upstream: skills/example\n" + " local: .github/skills/example\n" + " canonical_name: example\n" + "watchlist: []\n", + encoding="utf-8", + ) + + with pytest.raises(ValueError, match="not-an-asset"): + load_managed_resources(manifest) diff --git a/tests/github/skills/local-agent-sync-external-resources/scripts/test_network_boundary.py b/tests/github/skills/local-agent-sync-external-resources/scripts/test_network_boundary.py new file mode 100644 index 00000000..6006ae44 --- /dev/null +++ b/tests/github/skills/local-agent-sync-external-resources/scripts/test_network_boundary.py @@ -0,0 +1,220 @@ +import ast +import sys +from pathlib import Path + +import pytest + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SCRIPT_DIR = REPO_ROOT / ".github/skills/local-agent-sync-external-resources/scripts" +sys.path.insert(0, SCRIPT_DIR.as_posix()) + + +_FORBIDDEN_COMMANDS = { + "pull", + "pip", + "uv", + "npm", + "brew", + "yarn", + "pnpm", +} + + +def _extract_subprocess_commands(script: Path) -> list[list[str]]: + source = script.read_text(encoding="utf-8") + tree = ast.parse(source) + commands: list[list[str]] = [] + + for node in ast.walk(tree): + if not isinstance(node, ast.Call): + continue + func = node.func + func_name = "" + if isinstance(func, ast.Attribute): + func_name = func.attr + elif isinstance(func, ast.Name): + func_name = func.id + if func_name not in ("run", "Popen", "check_output", "check_call"): + continue + + all_args: list[str] = [] + if node.args: + first = node.args[0] + if isinstance(first, ast.List): + for elt in first.elts: + if isinstance(elt, ast.Constant) and isinstance(elt.value, str): + all_args.append(elt.value) + elif isinstance(first, ast.Constant) and isinstance(first.value, str): + all_args.append(first.value) + + for kw in node.keywords: + if kw.arg == "args" and isinstance(kw.value, ast.List): + for elt in kw.value.elts: + if isinstance(elt, ast.Constant) and isinstance(elt.value, str): + all_args.append(elt.value) + + if all_args: + commands.append(all_args) + + return commands + + +def _find_argumentless_fetch(commands: list[list[str]]) -> list[list[str]]: + violations: list[list[str]] = [] + for cmd in commands: + if not cmd: + continue + base = cmd[0] + if base != "git": + continue + if "fetch" not in cmd: + continue + non_flag_args = [ + a + for a in cmd[1:] + if not a.startswith("-") + and not a.startswith("--") + and a != "fetch" + and not a.startswith("+") + ] + if not non_flag_args: + violations.append(cmd) + return violations + + +def test_no_script_invokes_argumentless_git_fetch() -> None: + for script in sorted(SCRIPT_DIR.glob("*.py")): + commands = _extract_subprocess_commands(script) + violations = _find_argumentless_fetch(commands) + assert not violations, ( + f"{script.name} invokes argumentless git fetch: {violations}" + ) + + +def test_no_script_invokes_forbidden_package_managers() -> None: + for script in sorted(SCRIPT_DIR.glob("*.py")): + commands = _extract_subprocess_commands(script) + for cmd in commands: + base = cmd[0] if cmd else "" + assert base not in _FORBIDDEN_COMMANDS, ( + f"{script.name} invokes forbidden command: {cmd}" + ) + + +def test_no_script_invokes_git_pull_or_remote_update() -> None: + for script in sorted(SCRIPT_DIR.glob("*.py")): + commands = _extract_subprocess_commands(script) + for cmd in commands: + if not cmd or cmd[0] != "git": + continue + if "pull" in cmd: + pytest.fail(f"{script.name} invokes git pull: {cmd}") + if len(cmd) >= 3 and cmd[1] == "remote" and cmd[2] == "update": + pytest.fail(f"{script.name} invokes git remote update: {cmd}") + + +def test_only_source_prepare_core_executes_git_fetch() -> None: + fetch_scripts: list[str] = [] + for script in sorted(SCRIPT_DIR.glob("*.py")): + source = script.read_text(encoding="utf-8") + if '"fetch"' in source or "'fetch'" in source: + tree = ast.parse(source) + has_subprocess = False + for node in ast.walk(tree): + if isinstance(node, (ast.Import, ast.ImportFrom)): + module = getattr(node, "module", "") or "" + if "subprocess" in module: + has_subprocess = True + break + for alias in getattr(node, "names", []): + if "subprocess" in alias.name: + has_subprocess = True + break + if not has_subprocess: + continue + has_git_command = False + for node in ast.walk(tree): + if isinstance(node, ast.Call): + func = node.func + func_name = "" + if isinstance(func, ast.Attribute): + func_name = func.attr + elif isinstance(func, ast.Name): + func_name = func.id + if func_name in ( + "run", + "Popen", + "check_output", + "check_call", + "_run_command", + ): + has_git_command = True + break + if has_git_command: + fetch_scripts.append(script.name) + + assert fetch_scripts == ["source_prepare_core.py"], ( + f"Expected only source_prepare_core.py to execute git fetch, " + f"found: {fetch_scripts}" + ) + + +def test_audit_does_not_call_prepare_sources() -> None: + source = (SCRIPT_DIR / "sync_external_resources.py").read_text(encoding="utf-8") + tree = ast.parse(source) + for node in ast.walk(tree): + if isinstance(node, ast.FunctionDef) and node.name == "_audit": + for inner in ast.walk(node): + if isinstance(inner, ast.Call): + if ( + isinstance(inner.func, ast.Name) + and inner.func.id == "prepare_sources" + ): + pytest.fail("_audit calls prepare_sources") + if ( + isinstance(inner.func, ast.Attribute) + and inner.func.attr == "prepare_sources" + ): + pytest.fail("_audit calls prepare_sources") + + +def test_plan_does_not_call_prepare_sources() -> None: + source = (SCRIPT_DIR / "sync_external_resources.py").read_text(encoding="utf-8") + tree = ast.parse(source) + for node in ast.walk(tree): + if isinstance(node, ast.FunctionDef) and node.name == "_plan": + for inner in ast.walk(node): + if isinstance(inner, ast.Call): + if ( + isinstance(inner.func, ast.Name) + and inner.func.id == "prepare_sources" + ): + pytest.fail("_plan calls prepare_sources") + if ( + isinstance(inner.func, ast.Attribute) + and inner.func.attr == "prepare_sources" + ): + pytest.fail("_plan calls prepare_sources") + + +def test_apply_does_not_call_prepare_sources() -> None: + source = (SCRIPT_DIR / "sync_external_resources.py").read_text(encoding="utf-8") + tree = ast.parse(source) + for node in ast.walk(tree): + if isinstance(node, ast.FunctionDef) and node.name == "_apply": + for inner in ast.walk(node): + if isinstance(inner, ast.Call): + if ( + isinstance(inner.func, ast.Name) + and inner.func.id == "prepare_sources" + ): + pytest.fail("_apply calls prepare_sources") + if ( + isinstance(inner.func, ast.Attribute) + and inner.func.attr == "prepare_sources" + ): + pytest.fail("_apply calls prepare_sources") diff --git a/tests/github/skills/local-agent-sync-external-resources/scripts/test_output.py b/tests/github/skills/local-agent-sync-external-resources/scripts/test_output.py new file mode 100644 index 00000000..3e566840 --- /dev/null +++ b/tests/github/skills/local-agent-sync-external-resources/scripts/test_output.py @@ -0,0 +1,68 @@ +import sys +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SCRIPT_DIR = REPO_ROOT / ".github/skills/local-agent-sync-external-resources/scripts" +sys.path.insert(0, SCRIPT_DIR.as_posix()) + +from sync_output_core import ( # noqa: E402 + OutputRecord, + escape_tsv, + render_tsv, +) + + +def test_escape_tsv_replaces_backslash_first() -> None: + assert escape_tsv("a\\b") == "a\\\\b" + + +def test_escape_tsv_replaces_tab() -> None: + assert escape_tsv("a\tb") == "a\\tb" + + +def test_escape_tsv_replaces_newline() -> None: + assert escape_tsv("a\nb") == "a\\nb" + + +def test_escape_tsv_replaces_carriage_return() -> None: + assert escape_tsv("a\rb") == "a\\rb" + + +def test_escape_tsv_order_backslash_before_others() -> None: + assert escape_tsv("\\\t\n\r") == "\\\\\\t\\n\\r" + + +def test_render_tsv_header_is_fixed() -> None: + records = [OutputRecord("summary", "mode", "ok", "audit")] + output = render_tsv(records) + first_line = output.split("\n", 1)[0] + assert first_line == "record\tkey\tstatus\tvalue" + + +def test_render_tsv_sorts_lexically_by_record_key_status_value() -> None: + records = [ + OutputRecord("z", "a", "ok", "v1"), + OutputRecord("a", "z", "ok", "v2"), + OutputRecord("a", "a", "ok", "v3"), + ] + output = render_tsv(records) + lines = output.strip().split("\n") + data_lines = lines[1:] + assert data_lines == [ + "a\ta\tok\tv3", + "a\tz\tok\tv2", + "z\ta\tok\tv1", + ] + + +def test_render_tsv_escapes_values() -> None: + records = [OutputRecord("summary", "note", "ok", "has\ttab")] + output = render_tsv(records) + data_lines = output.strip().split("\n")[1:] + assert len(data_lines) == 1 + assert "has\\ttab" in data_lines[0] + assert "\t" not in data_lines[0].split("\t")[3] diff --git a/tests/github/skills/local-agent-sync-external-resources/scripts/test_source_prepare.py b/tests/github/skills/local-agent-sync-external-resources/scripts/test_source_prepare.py new file mode 100644 index 00000000..a294928d --- /dev/null +++ b/tests/github/skills/local-agent-sync-external-resources/scripts/test_source_prepare.py @@ -0,0 +1,540 @@ +from __future__ import annotations + +import hashlib +import io +import os +import stat +import subprocess +import sys +import tarfile +from pathlib import Path + +import pytest + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SCRIPT_DIR = REPO_ROOT / ".github/skills/local-agent-sync-external-resources/scripts" +sys.path.insert(0, SCRIPT_DIR.as_posix()) + +from source_prepare_core import ( # noqa: E402 + _build_fetch_command, + _cache_key_for_repository, + _extract_archive, + _validate_upstream_paths, + prepare_sources, +) +from sync_external_resources_core import ( # noqa: E402 + ManagedAsset, + ManagedResources, + ManagedSource, + validate_prepared_sources, +) + +_FULL_SHA40 = "a" * 40 + + +def _run_git(cwd: Path, args: list[str]) -> None: + subprocess.run( + ["git", *args], + cwd=cwd, + check=True, + capture_output=True, + text=True, + ) + + +def _commit_all(repo: Path) -> str: + _run_git(repo, ["add", "-A"]) + _run_git(repo, ["commit", "-m", "snapshot", "--allow-empty"]) + result = subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=repo, + capture_output=True, + text=True, + check=True, + ) + return result.stdout.strip() + + +def test_cache_key_is_deterministic_sha256() -> None: + url = "https://github.com/example/repo.git" + key = _cache_key_for_repository(url) + expected = hashlib.sha256(url.encode("utf-8")).hexdigest() + assert key == expected + + +def test_cache_key_shared_across_source_ids() -> None: + url = "https://github.com/openai/skills.git" + assert _cache_key_for_repository(url) == _cache_key_for_repository(url) + + +def test_fetch_command_includes_required_flags() -> None: + sha = "a" * 40 + cmd = _build_fetch_command(sha) + cmd_str = " ".join(cmd) + assert "--filter=blob:none" in cmd_str + assert "--no-tags" in cmd_str + assert "--no-recurse-submodules" in cmd_str + assert "--no-write-fetch-head" in cmd_str + assert "--refmap=" in cmd_str + assert "-c" in cmd + idx = cmd.index("-c") + assert cmd[idx + 1] == "fetch.fsckObjects=true" + + +@pytest.fixture +def fixture_remote(tmp_path: Path) -> tuple[Path, str]: + remote = tmp_path / "remote.git" + remote.mkdir() + _run_git(remote, ["init", "--bare"]) + _run_git(remote, ["config", "uploadpack.allowReachableSHA1InWant", "true"]) + _run_git(remote, ["config", "uploadpack.allowFilter", "true"]) + + work = tmp_path / "work" + work.mkdir() + _run_git(work, ["init"]) + _run_git(work, ["config", "user.email", "test@test.com"]) + _run_git(work, ["config", "user.name", "Test"]) + + skill_dir = work / "skills" / "target-skill" + skill_dir.mkdir(parents=True) + (skill_dir / "SKILL.md").write_text( + "---\nname: target-skill\n---\nTarget content.\n", + encoding="utf-8", + ) + + script = skill_dir / "run.sh" + script.write_text("#!/bin/sh\necho hello\n", encoding="utf-8") + script.chmod(script.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + + link_target = skill_dir / "inner.txt" + link_target.write_text("inner\n", encoding="utf-8") + link = skill_dir / "link-to-inner" + link.symlink_to("inner.txt") + + decoy = work / "decoy-8mib.bin" + decoy.write_bytes(b"\x00" * (8 * 1024 * 1024)) + + other_skill = work / "skills" / "other-skill" / "SKILL.md" + other_skill.parent.mkdir(parents=True) + other_skill.write_text("---\nname: other\n---\n", encoding="utf-8") + + commit_sha = _commit_all(work) + _run_git(work, ["remote", "add", "origin", str(remote)]) + _run_git(work, ["push", "origin", "HEAD:refs/heads/main"]) + + return remote, commit_sha + + +def test_prepare_sources_cold_fetch_creates_selective_snapshot( + tmp_path: Path, + fixture_remote: tuple[Path, str], +) -> None: + remote_path, commit_sha = fixture_remote + workspace = tmp_path / "workspace" + workspace.mkdir() + sources_root = tmp_path / "sources" + sources_root.mkdir() + + asset = ManagedAsset( + source="test-source", + upstream="skills/target-skill", + local=".github/skills/target-skill", + canonical_name="target-skill", + ) + source = ManagedSource( + source_id="test-source", + repository=str(remote_path), + ref=commit_sha, + advertised_ref=None, + assets=(asset,), + ) + resources = ManagedResources( + sources=(source,), + replacements=(), + watchlist=(), + ) + + results = prepare_sources(resources, workspace, sources_root) + + assert len(results) == 1 + result = results[0] + assert result.source_id == "test-source" + assert result.ref == commit_sha + assert result.cache_status in ("fetched", "cached") + assert result.materialized_files >= 3 + assert result.materialized_bytes > 0 + + snapshot = sources_root / "test-source" + assert snapshot.exists() + + skill_md = snapshot / "skills" / "target-skill" / "SKILL.md" + assert skill_md.exists() + assert "Target content." in skill_md.read_text(encoding="utf-8") + + run_sh = snapshot / "skills" / "target-skill" / "run.sh" + assert run_sh.exists() + assert os.access(run_sh, os.X_OK) + + link = snapshot / "skills" / "target-skill" / "link-to-inner" + assert link.is_symlink() + assert link.read_text(encoding="utf-8") == "inner\n" + + assert not (snapshot / ".git").exists() + assert not (snapshot / "decoy-8mib.bin").exists() + assert not (snapshot / "skills" / "other-skill").exists() + + +def test_prepare_sources_warm_run_is_cached( + tmp_path: Path, + fixture_remote: tuple[Path, str], +) -> None: + remote_path, commit_sha = fixture_remote + workspace = tmp_path / "workspace" + workspace.mkdir() + sources_root = tmp_path / "sources" + sources_root.mkdir() + + asset = ManagedAsset( + source="test-source", + upstream="skills/target-skill", + local=".github/skills/target-skill", + canonical_name="target-skill", + ) + source = ManagedSource( + source_id="test-source", + repository=str(remote_path), + ref=commit_sha, + advertised_ref=None, + assets=(asset,), + ) + resources = ManagedResources( + sources=(source,), + replacements=(), + watchlist=(), + ) + + first = prepare_sources(resources, workspace, sources_root) + second = prepare_sources(resources, workspace, sources_root) + + assert second[0].cache_status == "cached" + assert second[0].cache_bytes_added == 0 + assert second[0].materialized_files == first[0].materialized_files + assert second[0].materialized_bytes == first[0].materialized_bytes + + +def test_prepare_sources_cold_metrics( + tmp_path: Path, + fixture_remote: tuple[Path, str], +) -> None: + remote_path, commit_sha = fixture_remote + workspace = tmp_path / "workspace" + workspace.mkdir() + sources_root = tmp_path / "sources" + sources_root.mkdir() + + asset = ManagedAsset( + source="test-source", + upstream="skills/target-skill", + local=".github/skills/target-skill", + canonical_name="target-skill", + ) + source = ManagedSource( + source_id="test-source", + repository=str(remote_path), + ref=commit_sha, + advertised_ref=None, + assets=(asset,), + ) + resources = ManagedResources( + sources=(source,), + replacements=(), + watchlist=(), + ) + + results = prepare_sources(resources, workspace, sources_root) + result = results[0] + + assert result.cache_status == "fetched" + assert result.materialized_bytes < 1 * 1024 * 1024 + assert result.cache_bytes_added > 0 + assert result.duration_ms >= 0 + + +def test_prepare_sources_warm_metrics( + tmp_path: Path, + fixture_remote: tuple[Path, str], +) -> None: + remote_path, commit_sha = fixture_remote + workspace = tmp_path / "workspace" + workspace.mkdir() + sources_root = tmp_path / "sources" + sources_root.mkdir() + + asset = ManagedAsset( + source="test-source", + upstream="skills/target-skill", + local=".github/skills/target-skill", + canonical_name="target-skill", + ) + source = ManagedSource( + source_id="test-source", + repository=str(remote_path), + ref=commit_sha, + advertised_ref=None, + assets=(asset,), + ) + resources = ManagedResources( + sources=(source,), + replacements=(), + watchlist=(), + ) + + first = prepare_sources(resources, workspace, sources_root) + second = prepare_sources(resources, workspace, sources_root) + + assert second[0].cache_status == "cached" + assert second[0].cache_bytes_added == 0 + assert second[0].materialized_files == first[0].materialized_files + assert second[0].materialized_bytes == first[0].materialized_bytes + + +def test_validate_upstream_paths_rejects_absolute() -> None: + with pytest.raises(ValueError): + _validate_upstream_paths(["/etc/passwd"]) + + +def test_validate_upstream_paths_rejects_empty() -> None: + with pytest.raises(ValueError): + _validate_upstream_paths([""]) + + +def test_validate_upstream_paths_rejects_dot() -> None: + with pytest.raises(ValueError): + _validate_upstream_paths(["."]) + + +def test_validate_upstream_paths_rejects_dotdot() -> None: + with pytest.raises(ValueError): + _validate_upstream_paths([".."]) + + +def test_validate_upstream_paths_rejects_backslash() -> None: + with pytest.raises(ValueError): + _validate_upstream_paths(["skills\\bad"]) + + +def test_validate_upstream_paths_rejects_duplicate() -> None: + with pytest.raises(ValueError): + _validate_upstream_paths(["skills/a", "skills/a"]) + + +def test_validate_upstream_paths_rejects_overlapping() -> None: + with pytest.raises(ValueError): + _validate_upstream_paths(["skills", "skills/sub"]) + + +def test_validate_prepared_sources_reports_missing_metadata( + tmp_path: Path, +) -> None: + asset = ManagedAsset( + source="test-source", + upstream="skills/example", + local=".github/skills/example", + canonical_name="example", + ) + source = ManagedSource( + source_id="test-source", + repository="https://example.com/repo.git", + ref="a" * 40, + advertised_ref=None, + assets=(asset,), + ) + resources = ManagedResources( + sources=(source,), + replacements=(), + watchlist=(), + ) + sources_root = tmp_path / "sources" + sources_root.mkdir() + + with pytest.raises(ValueError, match="test-source"): + validate_prepared_sources(resources, sources_root) + + +def _tar_bytes_with(info: tarfile.TarInfo, payload: bytes = b"") -> bytes: + buffer = io.BytesIO() + with tarfile.open(fileobj=buffer, mode="w") as tar: + info.size = len(payload) + tar.addfile(info, io.BytesIO(payload)) + return buffer.getvalue() + + +def test_extract_archive_rejects_parent_traversal_member(tmp_path: Path) -> None: + export_dir = tmp_path / "export" + export_dir.mkdir() + archive = _tar_bytes_with(tarfile.TarInfo("../escaped.txt"), b"escaped\n") + + with pytest.raises(tarfile.FilterError): + _extract_archive(archive, export_dir) + + assert not (tmp_path / "escaped.txt").exists() + + +def test_extract_archive_rejects_absolute_member(tmp_path: Path) -> None: + export_dir = tmp_path / "export" + export_dir.mkdir() + archive = _tar_bytes_with(tarfile.TarInfo("/tmp/escaped.txt"), b"escaped\n") + + with pytest.raises(tarfile.FilterError): + _extract_archive(archive, export_dir) + + +def test_extract_archive_rejects_symlink_escaping_snapshot(tmp_path: Path) -> None: + export_dir = tmp_path / "export" + export_dir.mkdir() + info = tarfile.TarInfo("link") + info.type = tarfile.SYMTYPE + info.linkname = "../../etc/passwd" + archive = _tar_bytes_with(info) + + with pytest.raises(tarfile.FilterError): + _extract_archive(archive, export_dir) + + +def test_advertised_ref_fallback_is_reported_as_advertised_ref( + tmp_path: Path, + fixture_remote: tuple[Path, str], + monkeypatch: pytest.MonkeyPatch, +) -> None: + remote_path, commit_sha = fixture_remote + workspace = tmp_path / "workspace" + workspace.mkdir() + sources_root = tmp_path / "sources" + sources_root.mkdir() + + asset = ManagedAsset( + source="strict-source", + upstream="skills/target-skill", + local=".github/skills/target-skill", + canonical_name="target-skill", + ) + source = ManagedSource( + source_id="strict-source", + repository=str(remote_path), + ref=commit_sha, + advertised_ref="refs/heads/main", + assets=(asset,), + ) + resources = ManagedResources(sources=(source,), replacements=(), watchlist=()) + + import source_prepare_core + + def failing_fetch_sha(cache: Path, sha: str) -> None: + from sync_external_resources_core import SyncCommandError + + raise SyncCommandError(["git", "fetch"], 128, "simulated failure") + + monkeypatch.setattr(source_prepare_core, "_fetch_sha", failing_fetch_sha) + + results = prepare_sources(resources, workspace, sources_root) + + assert results[0].cache_status == "fetched" + assert results[0].fetch_strategy == "advertised-ref" + assert ( + sources_root / "strict-source" / "skills" / "target-skill" / "SKILL.md" + ).exists() + + +def _single_source_resources(remote_path: Path, commit_sha: str) -> ManagedResources: + asset = ManagedAsset( + source="test-source", + upstream="skills/target-skill", + local=".github/skills/target-skill", + canonical_name="target-skill", + ) + source = ManagedSource( + source_id="test-source", + repository=str(remote_path), + ref=commit_sha, + advertised_ref=None, + assets=(asset,), + ) + return ManagedResources(sources=(source,), replacements=(), watchlist=()) + + +def test_rebuild_cache_refetches_even_when_pin_is_present( + tmp_path: Path, + fixture_remote: tuple[Path, str], +) -> None: + remote_path, commit_sha = fixture_remote + workspace = tmp_path / "workspace" + workspace.mkdir() + sources_root = tmp_path / "sources" + sources_root.mkdir() + resources = _single_source_resources(remote_path, commit_sha) + + warm = prepare_sources(resources, workspace, sources_root) + assert warm[0].cache_status == "fetched" + + rebuilt = prepare_sources(resources, workspace, sources_root, rebuild_cache=True) + + assert rebuilt[0].cache_status == "rebuilt" + assert rebuilt[0].fetch_strategy == "direct-sha" + assert rebuilt[0].cache_bytes_added > 0 + assert ( + sources_root / "test-source" / "skills" / "target-skill" / "SKILL.md" + ).exists() + + +def test_rebuild_cache_leaves_no_staging_directories( + tmp_path: Path, + fixture_remote: tuple[Path, str], +) -> None: + remote_path, commit_sha = fixture_remote + workspace = tmp_path / "workspace" + workspace.mkdir() + sources_root = tmp_path / "sources" + sources_root.mkdir() + resources = _single_source_resources(remote_path, commit_sha) + + prepare_sources(resources, workspace, sources_root) + prepare_sources(resources, workspace, sources_root, rebuild_cache=True) + + repositories = workspace / "cache" / "repositories" + leftovers = [ + entry.name + for entry in repositories.iterdir() + if entry.name.endswith(".rebuild") or entry.name.endswith(".prior") + ] + assert leftovers == [] + + +import source_prepare_core # noqa: E402 + + +def test_network_fetch_uses_extended_timeout(monkeypatch: pytest.MonkeyPatch) -> None: + calls: list[tuple[list[str], int]] = [] + + def fake_run_command( + command: list[str], cwd: Path | None = None, timeout: int = 60 + ): + calls.append((command, timeout)) + return subprocess.CompletedProcess(command, 0, "", "") + + monkeypatch.setattr(source_prepare_core, "_run_command", fake_run_command) + + source_prepare_core._fetch_sha(Path("/nonexistent-cache"), _FULL_SHA40) + source_prepare_core._fetch_advertised_ref( + Path("/nonexistent-cache"), "refs/heads/main" + ) + + assert source_prepare_core.NETWORK_COMMAND_TIMEOUT_SECONDS >= 900 + assert all( + timeout == source_prepare_core.NETWORK_COMMAND_TIMEOUT_SECONDS + for _, timeout in calls + ) + assert len(calls) == 2 diff --git a/tests/github/skills/local-agent-sync-external-resources/test_external_resource_catalog_contract.py b/tests/github/skills/local-agent-sync-external-resources/test_external_resource_catalog_contract.py index 30f1232a..d2b76df4 100644 --- a/tests/github/skills/local-agent-sync-external-resources/test_external_resource_catalog_contract.py +++ b/tests/github/skills/local-agent-sync-external-resources/test_external_resource_catalog_contract.py @@ -26,7 +26,7 @@ def _write_valid_managed_resources(root: Path) -> None: sources: obra-superpowers: repository: https://github.com/obra/superpowers.git - ref: abc123 + ref: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa assets: - upstream: skills/brainstorming local: .github/skills/superpowers-brainstorming @@ -67,14 +67,14 @@ def test_catalog_check_rejects_duplicate_managed_target(tmp_path: Path) -> None: sources: source-a: repository: https://example.com/a.git - ref: abc + ref: aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa assets: - upstream: skills/one local: .github/skills/same canonical_name: same source-b: repository: https://example.com/b.git - ref: def + ref: bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb assets: - upstream: skills/two local: .github/skills/same diff --git a/tests/github/skills/local-agent-sync-install-ai-resources/scripts/test_apply_paths.py b/tests/github/skills/local-agent-sync-install-ai-resources/scripts/test_apply_paths.py index 66030165..ad0181c3 100644 --- a/tests/github/skills/local-agent-sync-install-ai-resources/scripts/test_apply_paths.py +++ b/tests/github/skills/local-agent-sync-install-ai-resources/scripts/test_apply_paths.py @@ -311,7 +311,7 @@ def test_temporary_home_sync_links_skills_preserves_home_only_and_copies_agents( assert payload["counts"]["residual"] == 0 -def test_agents_md_sync_removes_repository_local_rules_and_overwrites_home_copy( +def test_agents_md_sync_keeps_optional_local_reference_and_overwrites_home_copy( tmp_path: Path, capsys ) -> None: refs = tmp_path / ".github/skills/local-agent-sync-install-ai-resources/references" @@ -354,17 +354,17 @@ def test_agents_md_sync_removes_repository_local_rules_and_overwrites_home_copy( source.write_text( """# Global agent policy -`` - Shared policy. -`` - -`` +If `AGENTS.local.md` exists next to this file, load and apply it after this +baseline. If it does not exist, continue without error. +""", + encoding="utf-8", + ) + (tmp_path / "AGENTS.local.md").write_text( + """# Repository-local policy Repository-only policy. - -`` """, encoding="utf-8", ) @@ -395,11 +395,10 @@ def test_agents_md_sync_removes_repository_local_rules_and_overwrites_home_copy( target.read_text(encoding="utf-8") == """# Global agent policy -`` - Shared policy. -`` +If `AGENTS.local.md` exists next to this file, load and apply it after this +baseline. If it does not exist, continue without error. """ ) manifest = json.loads( @@ -411,6 +410,83 @@ def test_agents_md_sync_removes_repository_local_rules_and_overwrites_home_copy( assert manifest["managed_resources"][0]["resource_family"] == "agents-md" +def test_agents_md_sync_accepts_missing_optional_local_policy( + tmp_path: Path, capsys +) -> None: + refs = tmp_path / ".github/skills/local-agent-sync-install-ai-resources/references" + refs.mkdir(parents=True) + (refs / "home-sync-catalog.yaml").write_text( + """version: 1 +defaults: + include_internal_skills: false + include_local_skills: false + include_unlisted_skills: false + unmanaged_existing_skills_policy: repo-wins + excluded_skills: [] + skill_targets: [] +resources: + - resource_id: global-agents + source_family: agents-md + source_path: AGENTS.md + include_targets: [agents.md] + target_support: documented + notes: Portable global agent baseline. +""", + encoding="utf-8", + ) + (refs / "runtime-support-matrix.yaml").write_text( + """version: 1 +rows: + - target: agents.md + resource_family: agents-md + support_level: Documented + home_path: ~/.agents/AGENTS.md + direct_copy_possible: true + translation_required: false + include_in_v1: true + evidence: [] + notes: Portable global agent baseline. +""", + encoding="utf-8", + ) + (tmp_path / "AGENTS.md").write_text( + """# Global agent policy + +Shared policy. + +If `AGENTS.local.md` exists next to this file, load and apply it after this +baseline. If it does not exist, continue without error. +""", + encoding="utf-8", + ) + home = tmp_path / "home" + (home / ".agents").mkdir(parents=True) + + assert ( + run( + parse_args( + [ + "sync", + "--source-root", + str(tmp_path), + "--home-root", + str(home), + "--targets", + "agents.md", + ] + ) + ) + == 0 + ) + capsys.readouterr() + + assert ( + (home / ".agents/AGENTS.md") + .read_text(encoding="utf-8") + .endswith("If it does not exist, continue without error.\n") + ) + + def test_copilot_agents_are_symlinked_with_write_through( tmp_path: Path, capsys ) -> None: diff --git a/tests/github/skills/local-agent-sync-install-ai-resources/scripts/test_contracts.py b/tests/github/skills/local-agent-sync-install-ai-resources/scripts/test_contracts.py index 6da1d6b6..d01dcd1b 100644 --- a/tests/github/skills/local-agent-sync-install-ai-resources/scripts/test_contracts.py +++ b/tests/github/skills/local-agent-sync-install-ai-resources/scripts/test_contracts.py @@ -67,6 +67,60 @@ def test_install_payload_reports_linked_and_unlinked_without_bisync() -> None: assert "bisync" not in compact +def test_live_catalog_discovers_all_non_local_agents_for_copilot() -> None: + resources = load_home_sync_catalog(REPO_ROOT) + + expected_paths = { + path.relative_to(REPO_ROOT).as_posix() + for path in (REPO_ROOT / ".github/agents").glob("*.agent.md") + if not path.name.startswith("local-") + } + agent_resources = { + resource.source_path + for resource in resources + if resource.source_family == "agents" + } + + assert agent_resources == expected_paths + assert all( + resource.include_targets == ("codex", "copilot", "opencode") + for resource in resources + if resource.source_family == "agents" + ) + assert not any( + resource.source_path.endswith("local-sync-external-resources.agent.md") + for resource in resources + ) + + +def test_agent_discovery_excludes_local_agents_and_keeps_runtime_targets( + tmp_path: Path, +) -> None: + references = ( + tmp_path / ".github/skills/local-agent-sync-install-ai-resources/references" + ) + references.mkdir(parents=True) + (references / "home-sync-catalog.yaml").write_text( + "version: 1\ndefaults:\n include_unlisted_skills: false\nresources: []\n", + encoding="utf-8", + ) + agents_root = tmp_path / ".github/agents" + agents_root.mkdir(parents=True) + (agents_root / "review.agent.md").write_text( + "---\nname: review\n---\n", encoding="utf-8" + ) + (agents_root / "local-review.agent.md").write_text( + "---\nname: local-review\n---\n", encoding="utf-8" + ) + + resources = load_home_sync_catalog(tmp_path) + + assert [resource.source_path for resource in resources] == [ + ".github/agents/review.agent.md" + ] + assert resources[0].include_targets == ("codex", "copilot", "opencode") + + def test_empty_manifest_defaults_to_schema_v2(tmp_path: Path) -> None: payload, error = load_manifest(tmp_path / "manifest.json") @@ -259,7 +313,7 @@ def test_translate_agent_for_codex_preserves_body_and_handoffs(tmp_path: Path) - description: Review changes carefully. handoffs: - label: Escalate - agent: internal-review-code + agent: internal-gateway-review-code prompt: Include the risky files. --- Main body instructions. @@ -275,7 +329,7 @@ def test_translate_agent_for_codex_preserves_body_and_handoffs(tmp_path: Path) - assert payload["name"] == "review-agent" assert "Main body instructions." in payload["developer_instructions"] assert "## Handoffs" in payload["developer_instructions"] - assert "internal-review-code" in payload["developer_instructions"] + assert "internal-gateway-review-code" in payload["developer_instructions"] def test_load_home_sync_catalog_autodiscovers_skills_and_honors_policy( diff --git a/tests/github/skills/local-sync-repos/scripts/test_sync_contract.py b/tests/github/skills/local-sync-repos/scripts/test_sync_contract.py new file mode 100644 index 00000000..989e9ecc --- /dev/null +++ b/tests/github/skills/local-sync-repos/scripts/test_sync_contract.py @@ -0,0 +1,243 @@ +import subprocess +import sys +from pathlib import Path + +import pytest + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +SCRIPT_DIR = REPO_ROOT / ".github/skills/local-sync-repos/scripts" +sys.path.insert(0, SCRIPT_DIR.as_posix()) + +from sync_contract import ( # noqa: E402 + MANAGED_COPY_PATHS, + SourceContractError, + build_plan, +) + +MANAGED_COPY_PATHS_EXPECTED = ( + "AGENTS.md", + ".python-version", + ".pre-commit-config.yaml", + ".editorconfig", + ".github/copilot-instructions.md", + ".github/workflows/_pre-commit.yml", +) + + +def _git_init(repo: Path) -> None: + subprocess.run(["git", "init"], cwd=repo, check=True, capture_output=True) + subprocess.run( + ["git", "-C", str(repo), "config", "user.email", "test@example.com"], + check=True, + capture_output=True, + ) + subprocess.run( + ["git", "-C", str(repo), "config", "user.name", "Test"], + check=True, + capture_output=True, + ) + + +def _populate_source(source: Path) -> None: + (source / "AGENTS.md").write_text("# agents\n", encoding="utf-8") + (source / ".python-version").write_text("3.13\n", encoding="utf-8") + (source / ".pre-commit-config.yaml").write_text("repos: []\n", encoding="utf-8") + (source / ".editorconfig").write_text("root = true\n", encoding="utf-8") + (source / ".github").mkdir(exist_ok=True) + (source / ".github" / "copilot-instructions.md").write_text( + "# copilot\n", encoding="utf-8" + ) + (source / ".github" / "workflows").mkdir(parents=True, exist_ok=True) + (source / ".github" / "workflows" / "_pre-commit.yml").write_text( + "name: pre-commit\n", encoding="utf-8" + ) + instructions = source / ".github" / "instructions" + instructions.mkdir(parents=True, exist_ok=True) + (instructions / "internal-python.instructions.md").write_text( + "# python\n", encoding="utf-8" + ) + + +@pytest.fixture() +def source_repo(tmp_path: Path) -> Path: + repo = tmp_path / "source" + repo.mkdir() + _git_init(repo) + _populate_source(repo) + subprocess.run( + ["git", "-C", str(repo), "add", "-A"], check=True, capture_output=True + ) + subprocess.run( + ["git", "-C", str(repo), "commit", "-m", "init", "--allow-empty"], + check=True, + capture_output=True, + ) + return repo + + +@pytest.fixture() +def target_repo(tmp_path: Path) -> Path: + repo = tmp_path / "target" + repo.mkdir() + _git_init(repo) + subprocess.run( + ["git", "-C", str(repo), "commit", "-m", "init", "--allow-empty"], + check=True, + capture_output=True, + ) + return repo + + +def test_managed_copy_paths_constant_matches_approved_scope() -> None: + assert MANAGED_COPY_PATHS == MANAGED_COPY_PATHS_EXPECTED + + +def test_build_plan_creates_only_approved_managed_paths( + source_repo: Path, target_repo: Path +) -> None: + plan = build_plan(source_repo, target_repo) + mutations = { + (item.action, item.path) for item in plan.operations if item.is_mutation + } + assert mutations == { + *(("create", path) for path in MANAGED_COPY_PATHS_EXPECTED), + ("create", ".github/instructions/internal-python.instructions.md"), + ("create", "AGENTS.local.md"), + } + + +def test_build_plan_updates_changed_managed_file( + source_repo: Path, target_repo: Path +) -> None: + (target_repo / ".editorconfig").write_text("target\n", encoding="utf-8") + plan = build_plan(source_repo, target_repo) + assert ("update", ".editorconfig") in { + (item.action, item.path) for item in plan.operations + } + + +def test_build_plan_preserves_local_instruction_and_deletes_other_target_only_instruction( + source_repo: Path, target_repo: Path +) -> None: + local_path = target_repo / ".github/instructions/local-team.instructions.md" + stale_path = target_repo / ".github/instructions/stale.instructions.md" + local_path.parent.mkdir(parents=True, exist_ok=True) + local_path.write_text("local\n", encoding="utf-8") + stale_path.write_text("stale\n", encoding="utf-8") + plan = build_plan(source_repo, target_repo) + assert ( + "preserve", + ".github/instructions/local-team.instructions.md", + ) in {(item.action, item.path) for item in plan.operations} + assert ( + "delete", + ".github/instructions/stale.instructions.md", + ) in {(item.action, item.path) for item in plan.operations} + + +def test_existing_agents_local_is_preserved_byte_for_byte( + source_repo: Path, target_repo: Path +) -> None: + local_policy = target_repo / "AGENTS.local.md" + local_policy.write_bytes(b"consumer-owned\n") + plan = build_plan(source_repo, target_repo) + assert ("preserve", "AGENTS.local.md") in { + (item.action, item.path) for item in plan.operations + } + + +def test_nested_source_instructions_are_discovered( + source_repo: Path, target_repo: Path +) -> None: + nested = source_repo / ".github" / "instructions" / "nested" + nested.mkdir(parents=True, exist_ok=True) + (nested / "deep.instructions.md").write_text("deep\n", encoding="utf-8") + plan = build_plan(source_repo, target_repo) + assert ( + "create", + ".github/instructions/nested/deep.instructions.md", + ) in {(item.action, item.path) for item in plan.operations} + + +def test_identical_files_produce_no_mutation( + source_repo: Path, target_repo: Path +) -> None: + for relative in MANAGED_COPY_PATHS_EXPECTED: + target_file = target_repo / relative + target_file.parent.mkdir(parents=True, exist_ok=True) + target_file.write_bytes((source_repo / relative).read_bytes()) + instruction_src = ( + source_repo / ".github" / "instructions" / "internal-python.instructions.md" + ) + instruction_tgt = ( + target_repo / ".github" / "instructions" / "internal-python.instructions.md" + ) + instruction_tgt.parent.mkdir(parents=True, exist_ok=True) + instruction_tgt.write_bytes(instruction_src.read_bytes()) + (target_repo / "AGENTS.local.md").write_text( + ( + REPO_ROOT / ".github/skills/local-sync-repos/templates/AGENTS.local.md" + ).read_text(), + encoding="utf-8", + ) + plan = build_plan(source_repo, target_repo) + mutations = [op for op in plan.operations if op.is_mutation] + assert mutations == [] + + +def test_missing_source_path_raises_source_contract_error( + target_repo: Path, tmp_path: Path +) -> None: + empty_source = tmp_path / "empty-source" + empty_source.mkdir() + _git_init(empty_source) + subprocess.run( + ["git", "-C", str(empty_source), "commit", "-m", "init", "--allow-empty"], + check=True, + capture_output=True, + ) + with pytest.raises(SourceContractError, match="missing required source path"): + build_plan(empty_source, target_repo) + + +def test_fingerprint_is_deterministic(source_repo: Path, target_repo: Path) -> None: + first = build_plan(source_repo, target_repo) + second = build_plan(source_repo, target_repo) + assert first.fingerprint == second.fingerprint + assert len(first.fingerprint) == 64 + + +def test_dirty_managed_overlap_is_reported( + source_repo: Path, target_repo: Path +) -> None: + (target_repo / ".editorconfig").write_text("dirty\n", encoding="utf-8") + subprocess.run( + ["git", "-C", str(target_repo), "add", "-A"], check=True, capture_output=True + ) + subprocess.run( + ["git", "-C", str(target_repo), "commit", "-m", "seed", "--allow-empty"], + check=True, + capture_output=True, + ) + (target_repo / ".editorconfig").write_text("dirty-again\n", encoding="utf-8") + plan = build_plan(source_repo, target_repo) + assert ".editorconfig" in plan.dirty_managed_overlap + + +def test_dirty_unrelated_path_is_non_blocking( + source_repo: Path, target_repo: Path +) -> None: + (target_repo / "unrelated.txt").write_text("dirty\n", encoding="utf-8") + plan = build_plan(source_repo, target_repo) + assert plan.dirty_managed_overlap == () + + +def test_same_source_and_target_is_rejected( + source_repo: Path, +) -> None: + with pytest.raises(SourceContractError, match="same directory"): + build_plan(source_repo, source_repo) diff --git a/tests/github/skills/local-sync-repos/scripts/test_sync_repos_cli.py b/tests/github/skills/local-sync-repos/scripts/test_sync_repos_cli.py new file mode 100644 index 00000000..8f15c275 --- /dev/null +++ b/tests/github/skills/local-sync-repos/scripts/test_sync_repos_cli.py @@ -0,0 +1,210 @@ +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) +CLI = REPO_ROOT / ".github/skills/local-sync-repos/scripts/sync_repos.py" + + +MANAGED_COPY_PATHS_EXPECTED = ( + "AGENTS.md", + ".python-version", + ".pre-commit-config.yaml", + ".editorconfig", + ".github/copilot-instructions.md", + ".github/workflows/_pre-commit.yml", +) + + +def _git_init(repo: Path) -> None: + subprocess.run(["git", "init"], cwd=repo, check=True, capture_output=True) + subprocess.run( + ["git", "-C", str(repo), "config", "user.email", "test@example.com"], + check=True, + capture_output=True, + ) + subprocess.run( + ["git", "-C", str(repo), "config", "user.name", "Test"], + check=True, + capture_output=True, + ) + + +def _populate_source(source: Path) -> None: + (source / "AGENTS.md").write_text("# agents\n", encoding="utf-8") + (source / ".python-version").write_text("3.13\n", encoding="utf-8") + (source / ".pre-commit-config.yaml").write_text("repos: []\n", encoding="utf-8") + (source / ".editorconfig").write_text("root = true\n", encoding="utf-8") + (source / ".github").mkdir(exist_ok=True) + (source / ".github" / "copilot-instructions.md").write_text( + "# copilot\n", encoding="utf-8" + ) + (source / ".github" / "workflows").mkdir(parents=True, exist_ok=True) + (source / ".github" / "workflows" / "_pre-commit.yml").write_text( + "name: pre-commit\n", encoding="utf-8" + ) + instructions = source / ".github" / "instructions" + instructions.mkdir(parents=True, exist_ok=True) + (instructions / "internal-python.instructions.md").write_text( + "# python\n", encoding="utf-8" + ) + + +@pytest.fixture() +def source_repo(tmp_path: Path) -> Path: + repo = tmp_path / "source" + repo.mkdir() + _git_init(repo) + _populate_source(repo) + subprocess.run( + ["git", "-C", str(repo), "add", "-A"], check=True, capture_output=True + ) + subprocess.run( + ["git", "-C", str(repo), "commit", "-m", "init", "--allow-empty"], + check=True, + capture_output=True, + ) + return repo + + +@pytest.fixture() +def target_repo(tmp_path: Path) -> Path: + repo = tmp_path / "target" + repo.mkdir() + _git_init(repo) + subprocess.run( + ["git", "-C", str(repo), "commit", "-m", "init", "--allow-empty"], + check=True, + capture_output=True, + ) + return repo + + +def _run_cli( + command: str, source: Path, target: Path, output_format: str = "compact" +) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [ + sys.executable, + CLI.as_posix(), + command, + "--source-root", + str(source), + "--target-repo", + str(target), + "--format", + output_format, + ], + capture_output=True, + text=True, + ) + + +def test_plan_writes_only_target_tmp_plan(source_repo: Path, target_repo: Path) -> None: + result = _run_cli("plan", source_repo, target_repo) + assert result.returncode == 0 + assert (target_repo / "tmp/local-sync-repos.plan.md").is_file() + assert not (target_repo / "AGENTS.md").exists() + + +def test_apply_requires_matching_saved_plan( + source_repo: Path, target_repo: Path +) -> None: + result = _run_cli("apply", source_repo, target_repo) + assert result.returncode == 1 + assert "missing-plan" in result.stderr + + +def test_apply_blocks_dirty_managed_overlap( + source_repo: Path, target_repo: Path +) -> None: + _run_cli("plan", source_repo, target_repo) + (target_repo / ".editorconfig").write_text("dirty\n", encoding="utf-8") + result = _run_cli("apply", source_repo, target_repo) + assert result.returncode == 1 + assert "dirty-managed-overlap" in result.stderr + + +def test_apply_converges_and_preserves_consumer_owned_files( + source_repo: Path, target_repo: Path +) -> None: + local_instruction = target_repo / ".github/instructions/local-team.instructions.md" + local_agents = target_repo / "AGENTS.local.md" + local_instruction.parent.mkdir(parents=True, exist_ok=True) + local_instruction.write_bytes(b"local instruction\n") + local_agents.write_bytes(b"local policy\n") + _run_cli("plan", source_repo, target_repo) + assert _run_cli("apply", source_repo, target_repo).returncode == 0 + second = _run_cli("plan", source_repo, target_repo, output_format="json") + payload = json.loads(second.stdout) + assert payload["managed_mutation_paths"] == [] + assert local_instruction.read_bytes() == b"local instruction\n" + assert local_agents.read_bytes() == b"local policy\n" + + +def test_apply_rejects_stale_plan_fingerprint( + source_repo: Path, target_repo: Path +) -> None: + _run_cli("plan", source_repo, target_repo) + (target_repo / ".editorconfig").write_text("changed\n", encoding="utf-8") + subprocess.run( + ["git", "-C", str(target_repo), "add", "-A"], check=True, capture_output=True + ) + subprocess.run( + ["git", "-C", str(target_repo), "commit", "-m", "drift"], + check=True, + capture_output=True, + ) + result = _run_cli("apply", source_repo, target_repo) + assert result.returncode == 1 + assert "stale-plan" in result.stderr + + +def test_apply_deletes_target_only_non_local_instruction( + source_repo: Path, target_repo: Path +) -> None: + stale = target_repo / ".github/instructions/stale.instructions.md" + stale.parent.mkdir(parents=True, exist_ok=True) + stale.write_text("stale\n", encoding="utf-8") + _run_cli("plan", source_repo, target_repo) + assert _run_cli("apply", source_repo, target_repo).returncode == 0 + assert not stale.exists() + + +def test_agents_local_is_create_once(source_repo: Path, target_repo: Path) -> None: + _run_cli("plan", source_repo, target_repo) + assert _run_cli("apply", source_repo, target_repo).returncode == 0 + first_content = (target_repo / "AGENTS.local.md").read_bytes() + (target_repo / "AGENTS.local.md").write_bytes(b"consumer-edit\n") + _run_cli("plan", source_repo, target_repo) + assert _run_cli("apply", source_repo, target_repo).returncode == 0 + assert (target_repo / "AGENTS.local.md").read_bytes() == b"consumer-edit\n" + assert first_content != b"consumer-edit\n" + + +def test_compact_format_reports_operation_counts( + source_repo: Path, target_repo: Path +) -> None: + result = _run_cli("plan", source_repo, target_repo, output_format="compact") + assert result.returncode == 0 + payload = json.loads(result.stdout) + assert "operation_counts" in payload + assert payload["operation_counts"]["total"] > 0 + assert "by_action" in payload["operation_counts"] + + +def test_converged_apply_removes_target_plan_file( + source_repo: Path, target_repo: Path +) -> None: + _run_cli("plan", source_repo, target_repo) + plan_file = target_repo / "tmp/local-sync-repos.plan.md" + assert plan_file.is_file() + assert _run_cli("apply", source_repo, target_repo).returncode == 0 + assert not plan_file.exists() diff --git a/tests/github/skills/local-sync-repos/test_catalog_contract.py b/tests/github/skills/local-sync-repos/test_catalog_contract.py new file mode 100644 index 00000000..d0ce656e --- /dev/null +++ b/tests/github/skills/local-sync-repos/test_catalog_contract.py @@ -0,0 +1,71 @@ +from pathlib import Path + +REPO_ROOT = next( + parent + for parent in Path(__file__).resolve().parents + if (parent / "AGENTS.md").exists() and (parent / ".github").exists() +) + + +RETIRED_IDENTIFIERS = ( + "local-sync-global-copilot-configs-into-repo", + "local-agent-sync-global-copilot-configs-into-repo", +) + +RETIRED_BUNDLE_PATHS = ( + ".github/agents/local-sync-global-copilot-configs-into-repo.agent.md", + ".github/skills/local-agent-sync-global-copilot-configs-into-repo/SKILL.md", + ".github/skills/local-agent-sync-global-copilot-configs-into-repo/agents/openai.yaml", + ".github/skills/local-agent-sync-global-copilot-configs-into-repo/references/sync-contract.md", + ".github/prompts/internal-sync-plan.prompt.md", + ".github/scripts/sync_copilot_catalog.py", + ".github/scripts/lib/syncing.py", +) + +NEW_BUNDLE_PATHS = ( + ".github/agents/local-sync-repos.agent.md", + ".github/skills/local-sync-repos/SKILL.md", + ".github/skills/local-sync-repos/scripts/sync_contract.py", + ".github/skills/local-sync-repos/scripts/sync_repos.py", + ".github/skills/local-sync-repos/references/sync-contract.md", + ".github/skills/local-sync-repos/templates/AGENTS.local.md", + ".github/skills/local-sync-repos/agents/openai.yaml", +) + + +def test_local_sync_repos_agent_points_to_one_core_skill() -> None: + text = (REPO_ROOT / ".github/agents/local-sync-repos.agent.md").read_text() + assert "name: local-sync-repos" in text + assert "- `local-sync-repos`" in text + assert "agents: []" in text + assert "## Output Expectations" in text + + +def test_retired_sync_identifiers_have_zero_matches() -> None: + matches: list[str] = [] + this_file = Path(__file__).resolve() + for path in REPO_ROOT.rglob("*"): + if ( + not path.is_file() + or ".git" in path.parts + or "tmp" in path.parts + or "graphify-out" in path.parts + or "__pycache__" in path.parts + ): + continue + if path.resolve() == this_file: + continue + text = path.read_text(encoding="utf-8", errors="ignore") + if any(identifier in text for identifier in RETIRED_IDENTIFIERS): + matches.append(path.relative_to(REPO_ROOT).as_posix()) + assert matches == [] + + +def test_retired_bundle_paths_are_absent() -> None: + missing = [p for p in RETIRED_BUNDLE_PATHS if (REPO_ROOT / p).exists()] + assert missing == [] + + +def test_new_bundle_paths_are_present() -> None: + missing = [p for p in NEW_BUNDLE_PATHS if not (REPO_ROOT / p).exists()] + assert missing == [] diff --git a/tests/test_repository_test_layout_contract.py b/tests/test_repository_test_layout_contract.py index 150a2fb0..7c35ddce 100644 --- a/tests/test_repository_test_layout_contract.py +++ b/tests/test_repository_test_layout_contract.py @@ -13,7 +13,8 @@ ANTI_PATTERNS = ( REPO_ROOT / ".github/skills/internal-python/references/review-anti-patterns.md" ) -SKILL_PATH_PATTERN = re.compile(r"\.github/skills/([^/]+)/") +SKILL_PATH_PATTERN = re.compile(r"\.github/skills/([a-z0-9-]+)(?:/|[\"'])") +AGENT_PATH_PATTERN = re.compile(r"\.github/agents/([a-z0-9-]+)\.agent\.md") def test_global_guidance_documents_generic_test_placement_rule() -> None: @@ -48,25 +49,27 @@ def test_github_owned_python_tests_make_owner_obvious() -> None: rel_path = test_path.relative_to(TESTS_ROOT) text = test_path.read_text(encoding="utf-8") - if ".github/scripts/run.sh" in text: - if "github" not in rel_path.parts or "scripts" not in rel_path.parts: - violations.append( - f"{rel_path} should make the .github/scripts owner obvious" - ) + owners = set(SKILL_PATH_PATTERN.findall(text)) + owners.update(AGENT_PATH_PATTERN.findall(text)) + skill_or_agent_owner_is_obvious = ( + "github" in rel_path.parts + and "skills" in rel_path.parts + and any(owner in rel_path.parts for owner in owners) + ) + script_owner_is_obvious = ( + ".github/scripts" in text + and "github" in rel_path.parts + and "scripts" in rel_path.parts + ) + if skill_or_agent_owner_is_obvious or script_owner_is_obvious: continue - skill_match = SKILL_PATH_PATTERN.search(text) - if skill_match is None: + if not owners: continue - skill_name = skill_match.group(1) - if ( - "github" not in rel_path.parts - or "skills" not in rel_path.parts - or skill_name not in rel_path.parts - ): - violations.append( - f"{rel_path} should make the .github/skills/{skill_name}/ owner obvious" - ) + owner_names = ", ".join(sorted(owners)) + violations.append( + f"{rel_path} should make one of the referenced owners obvious: {owner_names}" + ) assert not violations, "\n".join(violations)