Compare commits
60 Commits
0f0ac5821f
...
v2.0.1
| Author | SHA1 | Date | |
|---|---|---|---|
| 68e08c2413 | |||
| d42f6368fe | |||
| c68e864159 | |||
| c7ba3d2ccf | |||
| 4d336bbf35 | |||
| 36596598ef | |||
| b1ea14df3e | |||
| de84d1b677 | |||
| 65bac15257 | |||
| b0ef503485 | |||
| bd2bf667c5 | |||
| ba7cec7672 | |||
| 56cc173f65 | |||
| b93af30750 | |||
| b9c7762463 | |||
| 1929ffd2da | |||
| 123ece2fb3 | |||
| 9385c77ac7 | |||
| 54d7bd80ba | |||
| 75a13c82f6 | |||
| 79c9089122 | |||
| ede3f06689 | |||
| b0d6d08239 | |||
| f7cc27908c | |||
| e7ebc667b3 | |||
| 64ffb9f35a | |||
| d02765d595 | |||
| 311e7cd22c | |||
| 2540e50fcc | |||
| a85bdbed42 | |||
| b6e68e9a2b | |||
| 76075223c7 | |||
| 36ba7a18f8 | |||
| 4a5c3c0cff | |||
| 1c6eababb0 | |||
| 2e395a4efa | |||
| f9b919d7e3 | |||
| c9fe2e8ab2 | |||
| ae178a95a2 | |||
| b4f5881973 | |||
| 099cf5846c | |||
| 3bfdf58960 | |||
| dee56c506a | |||
| 2e8732a8e5 | |||
| c3ec5f2d3d | |||
| a4a075b07e | |||
| b8dc400365 | |||
| cf625229f7 | |||
| a155af6827 | |||
| c16ec2d45a | |||
| f4bb1cf4e5 | |||
| a700b3771c | |||
| aa15fc850c | |||
| 874bf06b18 | |||
| 430f46b8e8 | |||
| 7ba3d9cf1d | |||
| 52bbd62286 | |||
| af085ed057 | |||
| 3f1ee47f1e | |||
| 9e612fd183 |
@@ -1,37 +1,38 @@
|
||||
{
|
||||
"name": "holocron",
|
||||
"description": "AI development skills for Claude Code and GitHub Copilot CLI — factory, design, implement, review, and cross-cutting workflows.",
|
||||
"version": "0.3.3",
|
||||
"version": "0.4.5",
|
||||
"owner": {
|
||||
"name": "Defame1297",
|
||||
"email": "defame1297@rkdr.net",
|
||||
"url": "https://git.dev.rkdr.net/Defame1297/"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "kyberforge",
|
||||
"description": "Skills and agents for creating, maintaining, and managing a Claude Code / Copilot CLI plugin marketplace.",
|
||||
"version": "1.4.1",
|
||||
"version": "1.6.0",
|
||||
"category": "Developer Tools",
|
||||
"source": "./plugins/kyberforge"
|
||||
},
|
||||
{
|
||||
"name": "bin",
|
||||
"description": "A place for things to be binned",
|
||||
"version": "1.1.1",
|
||||
"description": "Skills for everyday AI-assisted development work that is not tied to a single tool, forge or language, and has not yet been split into a focused plugin.",
|
||||
"version": "1.1.5",
|
||||
"category": "Utilities",
|
||||
"source": "./plugins/bin"
|
||||
},
|
||||
{
|
||||
"name": "git",
|
||||
"description": "Skills for working with Git — conventional commits, branch management, pull requests, and feature flow.",
|
||||
"version": "1.3.2",
|
||||
"description": "Skills and agents for working with a local Git clone over the git wire protocol, and for authoring and running the pre-commit hooks that guard it.",
|
||||
"version": "1.3.5",
|
||||
"category": "Version Control",
|
||||
"source": "./plugins/git"
|
||||
},
|
||||
{
|
||||
"name": "gitea",
|
||||
"description": "Skills for managing Gitea repositories — issues, pull requests, milestones, releases, and wikis.",
|
||||
"version": "1.3.3",
|
||||
"description": "Skills and agents for working with a Gitea forge through its HTTP API — the forge's own objects, as distinct from the local git clone.",
|
||||
"version": "1.3.6",
|
||||
"category": "Version Control",
|
||||
"source": "./plugins/gitea"
|
||||
},
|
||||
@@ -45,6 +46,7 @@
|
||||
{
|
||||
"name": "mattpocock-skills",
|
||||
"description": "Skills for Real Engineers — planning, TDD, architecture, and debugging workflows from Matt Pocock's .claude directory.",
|
||||
"version": "1.2.3",
|
||||
"category": "Productivity",
|
||||
"source": {
|
||||
"source": "github",
|
||||
@@ -57,7 +59,7 @@
|
||||
{
|
||||
"name": "lint",
|
||||
"description": "Skills and agents for configuring and running linters.",
|
||||
"version": "1.1.5",
|
||||
"version": "1.1.6",
|
||||
"category": "Developer Tools",
|
||||
"source": "./plugins/lint"
|
||||
}
|
||||
|
||||
@@ -1,13 +1,16 @@
|
||||
{
|
||||
"enabledPlugins": {
|
||||
"bin@holocron": true,
|
||||
"core@holocron": true,
|
||||
"git@holocron": true,
|
||||
"gitea@holocron": true,
|
||||
"kyberforge@holocron": true,
|
||||
"lint@holocron": true
|
||||
},
|
||||
"hooks": {
|
||||
"PreToolUse": []
|
||||
"SessionStart": [
|
||||
{
|
||||
"matcher": "startup",
|
||||
"hooks": [
|
||||
{
|
||||
"type": "command",
|
||||
"command": "\"${CLAUDE_PROJECT_DIR}/.claude/hooks/kyberforge/.apm/hooks/check-apm-current.sh\"",
|
||||
"timeout": 380
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
20
.github/plugin/marketplace.json
vendored
20
.github/plugin/marketplace.json
vendored
@@ -1,37 +1,38 @@
|
||||
{
|
||||
"name": "holocron",
|
||||
"description": "AI development skills for Claude Code and GitHub Copilot CLI — factory, design, implement, review, and cross-cutting workflows.",
|
||||
"version": "0.3.3",
|
||||
"version": "0.4.5",
|
||||
"owner": {
|
||||
"name": "Defame1297",
|
||||
"email": "defame1297@rkdr.net",
|
||||
"url": "https://git.dev.rkdr.net/Defame1297/"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "kyberforge",
|
||||
"description": "Skills and agents for creating, maintaining, and managing a Claude Code / Copilot CLI plugin marketplace.",
|
||||
"version": "1.4.1",
|
||||
"version": "1.6.0",
|
||||
"category": "Developer Tools",
|
||||
"source": "./plugins/kyberforge"
|
||||
},
|
||||
{
|
||||
"name": "bin",
|
||||
"description": "A place for things to be binned",
|
||||
"version": "1.1.1",
|
||||
"description": "Skills for everyday AI-assisted development work that is not tied to a single tool, forge or language, and has not yet been split into a focused plugin.",
|
||||
"version": "1.1.5",
|
||||
"category": "Utilities",
|
||||
"source": "./plugins/bin"
|
||||
},
|
||||
{
|
||||
"name": "git",
|
||||
"description": "Skills for working with Git — conventional commits, branch management, pull requests, and feature flow.",
|
||||
"version": "1.3.2",
|
||||
"description": "Skills and agents for working with a local Git clone over the git wire protocol, and for authoring and running the pre-commit hooks that guard it.",
|
||||
"version": "1.3.5",
|
||||
"category": "Version Control",
|
||||
"source": "./plugins/git"
|
||||
},
|
||||
{
|
||||
"name": "gitea",
|
||||
"description": "Skills for managing Gitea repositories — issues, pull requests, milestones, releases, and wikis.",
|
||||
"version": "1.3.3",
|
||||
"description": "Skills and agents for working with a Gitea forge through its HTTP API — the forge's own objects, as distinct from the local git clone.",
|
||||
"version": "1.3.6",
|
||||
"category": "Version Control",
|
||||
"source": "./plugins/gitea"
|
||||
},
|
||||
@@ -45,6 +46,7 @@
|
||||
{
|
||||
"name": "mattpocock-skills",
|
||||
"description": "Skills for Real Engineers — planning, TDD, architecture, and debugging workflows from Matt Pocock's .claude directory.",
|
||||
"version": "1.2.3",
|
||||
"category": "Productivity",
|
||||
"source": {
|
||||
"source": "github",
|
||||
@@ -57,7 +59,7 @@
|
||||
{
|
||||
"name": "lint",
|
||||
"description": "Skills and agents for configuring and running linters.",
|
||||
"version": "1.1.5",
|
||||
"version": "1.1.6",
|
||||
"category": "Developer Tools",
|
||||
"source": "./plugins/lint"
|
||||
}
|
||||
|
||||
28
.gitignore
vendored
28
.gitignore
vendored
@@ -24,3 +24,31 @@ node_modules/
|
||||
|
||||
# Claude Code local settings (machine-specific)
|
||||
.claude/settings.local.json
|
||||
|
||||
# APM dependencies
|
||||
apm_modules/
|
||||
|
||||
# APM install output — deployed copies of released plugin content, regenerated
|
||||
# by `apm install`. The authoring source is plugins/<name>/.apm/; committing a
|
||||
# deployed copy would add a third mirror of the same skills to drift against.
|
||||
.claude/skills/
|
||||
.claude/agents/
|
||||
|
||||
# APM hook deployment output — `apm install` copies each package's referenced
|
||||
# hook scripts here and tracks its own settings.json entries in the sidecar.
|
||||
# Regenerated on every install; the authoring source is
|
||||
# plugins/<name>/.apm/hooks/ (ADR-0019).
|
||||
.claude/hooks/
|
||||
.claude/apm-hooks.json
|
||||
|
||||
# `apm pack` bundle output. The pre-push gate runs pack with --dry-run, so this
|
||||
# only appears after a bare `apm pack` during a release; it is not repo content.
|
||||
build/
|
||||
|
||||
# `apm pack`'s manifest for the *root* package. Emitted beside the marketplace
|
||||
# manifest by a bare `apm pack`, and never tracked on any branch — the repo's
|
||||
# own paths hide it, since sync-plugin-content.sh redirects `apm pack -o` to a
|
||||
# scratch tree and the apm-pack-check-clean pre-push hook runs --dry-run. Scoped
|
||||
# to the file, not the directory: the sibling .claude-plugin/marketplace.json is
|
||||
# compiled output that IS committed and must stay tracked.
|
||||
/.claude-plugin/plugin.json
|
||||
|
||||
12
.mcp.json
Normal file
12
.mcp.json
Normal file
@@ -0,0 +1,12 @@
|
||||
{
|
||||
"mcpServers": {
|
||||
"obsidian": {
|
||||
"args": [
|
||||
"@bitbonsai/mcpvault@0.15.0",
|
||||
"docs/"
|
||||
],
|
||||
"command": "npx",
|
||||
"type": "stdio"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -28,7 +28,29 @@ repos:
|
||||
- id: pretty-format-json
|
||||
stages: ['pre-commit']
|
||||
args: [--autofix]
|
||||
exclude: '(^|/)(\.claude-plugin/plugin\.json|\.github/plugin/plugin\.json|\.claude-plugin/marketplace\.json|\.github/plugin/marketplace\.json)$|^\.agents/plugins/marketplace\.json$'
|
||||
# Every generated manifest lives at a KNOWN path, so every alternative is
|
||||
# root-anchored and spells that path out. This was five `(^|/)`
|
||||
# any-depth alternatives plus one `^` root-only one -- a mixture with no
|
||||
# rationale, under which a fixture or vendored tree containing
|
||||
# `.../.claude-plugin/plugin.json` would have been silently excluded from
|
||||
# formatting while an equivalent `.../.agents/plugins/marketplace.json`
|
||||
# would not. All fifteen real files (3 root marketplace manifests, 2 per
|
||||
# plugin x 6 plugins) match; anything else is hand-authored and gets
|
||||
# formatted.
|
||||
#
|
||||
# `.claude/settings.json` is the sixteenth, and it is excluded for a
|
||||
# different reason: apm OWNS that file (ADR-0018, ADR-0019), and
|
||||
# `apm audit --ci` replays the install into a scratch tree and diffs
|
||||
# the result byte-for-byte. `pretty-format-json` sorts object keys
|
||||
# unless `--no-sort-keys` is passed, while apm's hook integrator emits
|
||||
# insertion order (`matcher` before `hooks`, `type` before `command`).
|
||||
# Formatting the file therefore rewrites apm's output into a form apm
|
||||
# would never produce, and the `apm-audit-ci` pre-push hook reports it
|
||||
# as permanent drift on a file with no git diff -- exactly what
|
||||
# happened when the SessionStart hook first landed in 2e395a4.
|
||||
# Re-running `apm install` fixes the file; leaving it in scope here
|
||||
# would re-break it on the very commit that carries the fix.
|
||||
exclude: '^(\.claude-plugin/marketplace\.json|\.agents/plugins/marketplace\.json|\.github/plugin/marketplace\.json|plugins/[^/]+/\.claude-plugin/plugin\.json|plugins/[^/]+/\.github/plugin/plugin\.json|\.claude/settings\.json)$'
|
||||
- id: check-yaml
|
||||
stages: ['pre-commit']
|
||||
- id: trailing-whitespace
|
||||
@@ -46,8 +68,8 @@ repos:
|
||||
hooks:
|
||||
- id: run-tests
|
||||
name: Run test suite
|
||||
description: Run all test-*.sh files and bats suite
|
||||
entry: bash tests/run-tests.sh
|
||||
description: Run all test-*.sh files and bats suite. --strict because a suite that exits 77 (SKIPPED) at pre-push means a documented dependency is missing on this machine, and pre-commit prints nothing for a passing hook -- without it the gate went green having verified 15 of 17 suites on a vale-less PATH, with the skip list swallowed. Ad-hoc `bash tests/run-tests.sh` still skips gracefully.
|
||||
entry: bash tests/run-tests.sh --strict
|
||||
language: system
|
||||
stages: [pre-push]
|
||||
pass_filenames: false
|
||||
@@ -80,6 +102,15 @@ repos:
|
||||
pass_filenames: false
|
||||
always_run: true
|
||||
|
||||
- id: check-executables-allow-sync
|
||||
name: Check executables allow key sync
|
||||
description: Verify root apm.yml's executables.allow key names kyberforge's actual version -- apm matches that key by exact "<package>#<version>" lookup, so a version bump on one side alone silently stops deploying kyberforge's hooks/ and bin/ and lets the apm install go stale (see ADR-0019)
|
||||
entry: bash scripts/check-executables-allow-sync.sh
|
||||
language: system
|
||||
stages: [pre-push]
|
||||
pass_filenames: false
|
||||
always_run: true
|
||||
|
||||
- id: apm-marketplace-check
|
||||
name: apm marketplace check
|
||||
description: Validate every marketplace.packages[] entry resolves, including network reachability of remote refs -- catches stale/unreachable remote package references that check-manifests.sh deliberately skips (local-source checks only)
|
||||
@@ -91,12 +122,66 @@ repos:
|
||||
|
||||
- id: apm-audit-ci
|
||||
name: apm audit --ci
|
||||
description: apm's own producer-side lockfile/policy/hidden-content integrity gate, per apm's documented recommended CI block (see docs/research/docs/microsoft-apm/testing-and-validation.md)
|
||||
entry: apm audit --ci
|
||||
description: Run apm's producer-side CI gate over the root manifest AND each of the six plugin packages. Verifies exactly two things per manifest -- apm.yml parses as a valid APM manifest (manifest-parse), and, if it declares dependencies, apm.lock.yaml exists and is consistent (lockfile-exists). It does NOT enforce an org policy and does NOT scan for hidden Unicode; see the comment below for why. Reference:plugins/kyberforge/.apm/skills/apm-workflow/references/audit.md
|
||||
entry: bash -c 'for d in . plugins/*/; do (cd "$d" && apm audit --ci) || { echo "apm audit --ci failed in $d" >&2; exit 1; }; done'
|
||||
language: system
|
||||
stages: [pre-push]
|
||||
pass_filenames: false
|
||||
always_run: true
|
||||
# The description above deliberately claims less than this hook's old one
|
||||
# did ("lockfile/policy/hidden-content integrity"), because two of those
|
||||
# three were never happening:
|
||||
#
|
||||
# * POLICY. `apm audit --ci` discovers an org policy from the git remote,
|
||||
# and apm's discovery only understands github.com and Azure DevOps.
|
||||
# This repo's remote is a self-hosted Gitea, so discovery resolves
|
||||
# nothing and the run prints `No org policy found at unknown;
|
||||
# enforcement skipped`. apm's own message suggests
|
||||
# `policy.fetch_failure_default=block` in apm.yml "to fail closed" --
|
||||
# that was tried on a scratch copy and REJECTED: it does not make the
|
||||
# check meaningful, it makes it permanently red. `apm audit --ci` then
|
||||
# exits 1 with `No org policy found at unknown
|
||||
# (policy.fetch_failure_default=block)` on every push, because there is
|
||||
# no org policy to find and no supported way for this remote to serve
|
||||
# one. A gate that can never go green is not a gate. Revisit if this
|
||||
# repo ever gains a policy source apm can actually reach.
|
||||
# * HIDDEN CONTENT. The hidden-Unicode scan is plain `apm audit`, not
|
||||
# `apm audit --ci` (the two are different modes, and --ci refuses to
|
||||
# combine with --file/--strip/--dry-run/PACKAGE). Plain `apm audit`
|
||||
# here reports `No apm.lock.yaml found -- nothing to scan` and exits 0,
|
||||
# so adding it would buy a second vacuous check, not coverage.
|
||||
#
|
||||
# What IS left is worth keeping, and is now run against seven manifests
|
||||
# instead of one. lockfile-exists is conditional -- it is vacuous while
|
||||
# every apm.yml declares `dependencies: {apm: [], mcp: []}`, and it arms
|
||||
# itself the moment one does not (verified: adding a git dependency to
|
||||
# plugins/lint/apm.yml fails with `apm.yml declares dependencies but
|
||||
# apm.lock.yaml is absent`). manifest-parse is unconditional and fires on
|
||||
# any malformed manifest (verified: a dependency entry missing its
|
||||
# git/path/registry field fails with `Cannot parse apm.yml`). Running the
|
||||
# six plugin packages is what makes either reachable for them at all --
|
||||
# the root-only invocation audits the marketplace manifest and nothing
|
||||
# else. Costs ~0.5s per package, needs no network (checked under
|
||||
# `unshare -rn`), so this does NOT join apm-marketplace-check and
|
||||
# apm-pack-check-clean on the offline SKIP= list.
|
||||
|
||||
- id: check-apm-agents-valid
|
||||
name: Validate real APM agent files
|
||||
description: Run agent-audit's validate.sh over every plugins/*/.apm/agents/*.agent.md file in this repo -- the artifacts it governs, not fixtures
|
||||
entry: bash scripts/check-apm-agents-valid.sh
|
||||
language: system
|
||||
stages: [pre-push]
|
||||
pass_filenames: false
|
||||
always_run: true
|
||||
# validate.sh was previously exercised only by check-scope-walkup-sync,
|
||||
# and only against synthetic mktemp fixtures -- it had never run against
|
||||
# the four agent files it governs. That is how ADR-0016 could be amended
|
||||
# to bless a `disallowedTools` frontmatter field while validate.sh's
|
||||
# allowlist still rejected it: the spec and its enforcer disagreed and
|
||||
# every gate stayed green. The expected file set is derived from
|
||||
# `git ls-files` (the pattern tests/run-bats.sh established) rather than
|
||||
# a hardcoded count, and discovering zero files is an error, not a pass.
|
||||
# Needs no network.
|
||||
|
||||
- id: apm-pack-check-clean
|
||||
name: apm pack --check-clean
|
||||
@@ -115,6 +200,16 @@ repos:
|
||||
stages: [pre-push]
|
||||
pass_filenames: false
|
||||
always_run: true
|
||||
# verbose so the DOWNGRADED run is audible. This hook can pass while
|
||||
# having verified strictly less than its name claims:
|
||||
# CHECK_VALE_STYLE_SYNC_ALLOW_MISSING_VALE=1 skips all six glob probes
|
||||
# and says so on a `passed (text-level only, vale unavailable)` line.
|
||||
# pre-commit prints nothing at all for a passing hook, so without this
|
||||
# the opt-out reinstated exactly the silent vacuous pass the script was
|
||||
# written to kill, one level up -- the run showed a bare `Passed` and
|
||||
# the documented instruction to read that summary line was impossible to
|
||||
# follow in the one situation the opt-out exists for. The script's clean
|
||||
# output is a single line, so this costs one line per push.
|
||||
|
||||
- id: check-scope-walkup-sync
|
||||
name: Check scope walk-up implementations agree
|
||||
@@ -173,12 +268,21 @@ repos:
|
||||
|
||||
- id: skill-size-check
|
||||
stages: ['pre-commit']
|
||||
name: SKILL.md size ceiling
|
||||
description: Enforce agentskills.io's 500-line/5,000-token SKILL.md size ceiling
|
||||
name: SKILL.md size and context-budget ceilings
|
||||
description: Enforce agentskills.io's 500-line/2,770-whole-file-word spec ceilings AND ADR-0020's context budget -- description 250 chars SUGGESTION / 400 FAIL, body-only 600 words SUGGESTION / 900 FAIL, and every boundary-clause routing target resolving to a real skill or agent under plugins/*/.apm/
|
||||
entry: scripts/skill-size-check.sh
|
||||
language: script
|
||||
files: '^plugins/[^/]+/\.apm/skills/[^/]+/SKILL\.md$'
|
||||
pass_filenames: true
|
||||
verbose: true
|
||||
# verbose so the SUGGESTION tier is audible. ADR-0020 depends on it:
|
||||
# "A ceiling does not produce an average ... The halving depends
|
||||
# entirely on the 250-character SUGGESTION tier being visible and
|
||||
# respected." pre-commit prints nothing at all for a passing hook, and
|
||||
# a SUGGESTION deliberately does not fail, so without verbose every
|
||||
# suggestion would be swallowed -- the exact invisibility ADR-0013
|
||||
# records for Vale warnings. Costs nothing on a clean file: the script
|
||||
# prints only findings.
|
||||
|
||||
- id: vale-audit-prefilter-skill
|
||||
stages: ['pre-commit']
|
||||
|
||||
@@ -13,8 +13,11 @@
|
||||
files: '(^|/)agents/[^/]+\.md$|\.agent\.md$'
|
||||
|
||||
- id: kyberforge-skill-size-check
|
||||
name: SKILL.md size ceiling
|
||||
description: Enforce agentskills.io's 500-line/5,000-token SKILL.md size ceiling
|
||||
name: SKILL.md size and context-budget ceilings
|
||||
description: Enforce agentskills.io's 500-line/2,770-whole-file-word spec ceilings plus ADR-0020's context budget (description 250 chars SUGGESTION / 400 FAIL, body-only 600 words SUGGESTION / 900 FAIL, resolvable boundary-clause routing targets)
|
||||
entry: scripts/skill-size-check.sh
|
||||
language: script
|
||||
files: '(^|/)SKILL\.md$'
|
||||
# verbose so the SUGGESTION tier reaches a human -- pre-commit prints
|
||||
# nothing for a passing hook, and a SUGGESTION deliberately does not fail.
|
||||
verbose: true
|
||||
|
||||
57
AGENTS.md
57
AGENTS.md
@@ -1,52 +1,59 @@
|
||||
# Working in this repo
|
||||
|
||||
This repo is the global AI development configuration repository — the authoritative source for agent definitions, skills, workflows, and prompts across all projects. Built as a homelab tool intended to scale to professional environments.
|
||||
The global AI development configuration repository — the authoritative source for agent definitions, skills, workflows, and prompts across all projects.
|
||||
|
||||
This file carries only what applies to **every** session. Setup, prerequisites, and test commands are in `README.md`; the reasoning behind each enforcement gate is in `docs/spec/gates.md`.
|
||||
|
||||
## Structure
|
||||
|
||||
- `plugins/` — installable plugin units; each is an apm package (`apm.yml` + `.apm/`) carrying skills, agents, hooks, MCP servers, and bundled assets; install separately via `claude plugin install <name>@holocron`
|
||||
- `providers/claude-code/` — Claude Code adapter (deployed to `~/.claude/` via `install.sh`)
|
||||
- `plugins/` — six installable plugin units, each an apm package (`apm.yml` + `.apm/`). Root `apm.yml` declares all six as `dependencies.apm`; `apm install` deploys them into `.claude/skills/` and `.claude/agents/`, both gitignored install output.
|
||||
- `providers/claude-code/` — Claude Code adapter, deployed to `~/.claude/` via `scripts/install.sh`.
|
||||
|
||||
## Edit `.apm/`, never the flat mirror
|
||||
|
||||
Inside a plugin, `plugins/<name>/.apm/` is the **only** hand-edited content source. Everything else in a plugin root is generated:
|
||||
`plugins/<name>/.apm/` is the only hand-edited source for plugin content. The flat `plugins/<name>/{skills,agents,commands,instructions,extensions}/` directories, the merged `plugins/<name>/hooks/hooks.json`, and both `plugin.json` manifests are generated — nothing marks them as generated, so check the path before you edit. An edit to the mirror is discarded by the next sync and reported as drift by the `check-plugin-content-sync` pre-push hook.
|
||||
|
||||
- `scripts/sync-plugin-content.sh` generates the flat `plugins/<name>/{skills,agents,commands,instructions,extensions}/` directories and the merged `plugins/<name>/hooks/hooks.json` (ADR-0017)
|
||||
- `apm pack` generates both per-plugin manifests — `plugins/<name>/.claude-plugin/plugin.json` and `plugins/<name>/.github/plugin/plugin.json` — and **two of the three** root marketplace manifests: `.claude-plugin/marketplace.json` (apm's `claude` output profile) and `.agents/plugins/marketplace.json` (its `codex` profile, a differently-shaped file) (ADR-0015)
|
||||
- `scripts/sync-marketplace-mirror.sh` generates the third, `.github/plugin/marketplace.json` — Copilot CLI's legacy manifest path. **No apm output profile targets it**: apm ships exactly two marketplace output profiles, `claude` and `codex` (documented in `plugins/kyberforge/.apm/skills/apm-workflow/references/marketplace.md`). The mirror is a byte-identical copy of `.claude-plugin/marketplace.json`, gated by the `check-marketplace-mirror-sync` pre-push hook. Do not expect `apm pack` to refresh it — that assumption is exactly the drift this pair exists to prevent
|
||||
Not everything in a plugin root is generated. `README.md`, `docs/`, `bin/`, `sources.md`, `.mcp.json` and per-plugin extras are hand-authored there with no `.apm/` source — edit those in place. The rule is per-path, not per-directory. But a file placed *inside* a mirrored directory is deleted on the next sync (`sync_dir` runs `rm -rf` before every copy), so plugin-root documentation goes in `docs/`, never in `hooks/` or `skills/`.
|
||||
|
||||
Nothing labels a generated file as generated — `plugins/kyberforge/skills/forge/SKILL.md` is byte-identical to its `.apm/` original, with no marker in either. Check the path before you edit. An edit to the mirror is discarded by the next sync and is reported as drift by the `check-plugin-content-sync` pre-push hook, which is the earliest anyone finds out. Details in `docs/spec/architecture.md`.
|
||||
Full model: `docs/spec/architecture.md`.
|
||||
|
||||
## Prefer plugin skills over raw shell
|
||||
|
||||
This repo dogfoods its own plugins. Before shelling out to git, gitea, or lint tooling directly, check whether an installed skill already owns the operation — it usually does:
|
||||
This repo dogfoods its own plugins. Before shelling out, check whether a skill already owns the operation — it usually does:
|
||||
|
||||
- Commits, branches, history, worktrees, remotes → `git:git-commits`, `git:git-branches`, `git:git-history`, `git:git-worktrees`, `git:git-remotes`
|
||||
- Pre-commit hook install/config/troubleshooting → `git:pc-run` / `git:pc-author`
|
||||
- Issues, PRs, labels, milestones → `gitea:gitea-issues`, `gitea:gitea-prs`, `gitea:gitea-labels-milestones`; also `gitea:gitea-branches`, `gitea:gitea-files`, `gitea:gitea-releases`, or `gitea:gitea-workflow` when the domain is ambiguous
|
||||
- Vale prose linting → `lint:vale-config` / `lint:vale-run`
|
||||
- This repo's own AGENTS.md → `core:agentsmd-author` / `core:agentsmd-audit`
|
||||
- Commits, branches, history, worktrees, remotes → `git-commits`, `git-branches`, `git-history`, `git-worktrees`, `git-remotes`
|
||||
- Pre-commit hook install/config/troubleshooting → `pc-run` / `pc-author`
|
||||
- Issues, PRs, labels, milestones → `gitea-issues`, `gitea-prs`, `gitea-labels-milestones`; also `gitea-branches`, `gitea-files`, `gitea-releases`, or `gitea-workflow` when the domain is ambiguous
|
||||
- Vale prose linting → `vale-config` / `vale-run`
|
||||
- This repo's own AGENTS.md → `agentsmd-author` / `agentsmd-audit`
|
||||
|
||||
Use the bare, **unnamespaced** names. That is what `apm install` deploys and the only form this repo's own install produces — a project skill has no plugin to prefix (ADR-0018). Whether the `<plugin>:` form (`gitea:gitea-prs`) also resolves depends on native plugin installs at user scope, outside this repo; write the bare name either way.
|
||||
|
||||
Fall back to raw shell only when no skill covers it.
|
||||
|
||||
## Setup and testing
|
||||
## Session rules
|
||||
|
||||
- Install git hooks via `git:pc-run`, wiring all three stages — this repo's `.pre-commit-config.yaml` has no `default_install_hook_types`, so a plain install silently skips `commit-msg` (Conventional Commits) and `pre-push` (the 12-hook gate described below).
|
||||
- Install the `apm` CLI — four pre-push hooks shell out to it: `apm-marketplace-check`, `apm-audit-ci`, `apm-pack-check-clean`, and `check-plugin-content-sync` (via `scripts/sync-plugin-content.sh`, which wraps `apm pack`). The first three are bare `apm …` hook entries, so without it the push dies with an unhelpful "command not found". Use `kyberforge:apm-install`, or `curl -sSL https://aka.ms/apm-unix | sh`; verify with `apm --version`.
|
||||
- Install `jq` — required by `scripts/check-manifests.sh` and `scripts/sync-plugin-content.sh`, both pre-push. These at least fail loudly (`Error: jq is required but not installed`).
|
||||
- Install the `vale` binary — required by the `vale-audit-prefilter-skill`/`-agent` pre-commit hooks. Their `files:` patterns are `.apm/`-scoped: `^plugins/[^/]+/\.apm/skills/[^/]+/SKILL\.md$` and `^plugins/[^/]+/\.apm/agents/[^/]+\.agent\.md$`. Only the authoring source triggers them — a `SKILL.md` in the generated mirror matches neither pattern, so prose findings surface only when you edit the file you are supposed to be editing. Without the binary the hooks fail with a bare "command not found" and no install pointer. `brew install vale` (macOS), `snap install vale` (Linux), `choco install vale` (Windows), or see https://vale.sh/docs/vale-cli/installation/. No `vale sync` needed — the `Kyberforge` styles are committed under `plugins/kyberforge/.apm/skills/{skill-audit,agent-audit}/assets/vale/styles/`, not downloaded packages (see ADR-0014).
|
||||
- Run `bash tests/run-tests.sh` before considering any change done — it runs every `test-*.sh` script in the repo plus the bats suite (`--bats-only` for just bats). First run auto-initializes the bats submodules; no manual `git submodule update` needed.
|
||||
- Pushing runs 12 repo-defined pre-push hooks, not just the test suite — `run-tests` and `check-manifests`, plus generated-content drift gates (`check-plugin-content-sync`, `check-marketplace-mirror-sync`, `check-vale-style-sync`, `check-scope-walkup-sync`), apm's own gates (`apm-marketplace-check`, `apm-audit-ci`, `apm-pack-check-clean`), host validators (`validate-plugins`, `validate-marketplace`, both needing the `claude` CLI), and `check-release-needed`. Run `pre-commit run --hook-stage pre-push --all-files` locally — one command, the whole gate. That command reports **14**, not 12: pre-commit's own `meta` hooks, `check-hooks-apply` and `check-useless-excludes`, declare no `stages:` and so run at every stage including this one.
|
||||
- `apm-marketplace-check` needs the network. It resolves every `marketplace.packages[]` entry including the remote `mattpocock-skills` ref, and it is `always_run`, so an unreachable network hard-fails the push. `--offline` is not an escape hatch — it still exits 1 on that entry (`No cached refs (offline)`). To push without a network, skip that one hook using pre-commit's own mechanism: `SKIP=apm-marketplace-check git push`. Skip that hook alone — it is the only one whose failure mode is "no network". Every other pre-push hook is a real local check, and adding it to `SKIP` disarms it silently.
|
||||
- Author commits with `git:git-commits` — it validates Conventional Commits (enforced at `commit-msg`) for you.
|
||||
- **Do not add repo-owned keys to `.claude/settings.json`.** apm treats it as its own deployed artifact and `apm audit --ci` replays the install and diffs, so anything apm would not have written is permanent drift that fails the `apm-audit-ci` pre-push hook. A hook you want here is authored in `plugins/<name>/.apm/hooks/` and deployed by apm, never hand-written into that file. The `SessionStart` entry already in it is exactly that: kyberforge authors it in `plugins/kyberforge/.apm/hooks/hooks.json` and apm merges it in, so it is apm's own output, it is what the replay expects, and it belongs in the commit — do not strip it (ADR-0019). Machine-specific settings go in the gitignored `.claude/settings.local.json`; shared enforcement goes in `.pre-commit-config.yaml`.
|
||||
- **`apm.lock.yaml` turning up modified is expected, not a bug.** kyberforge's `SessionStart` hook runs `apm outdated` at startup and `apm update --yes` when something is behind, which rewrites the lock. Commit or discard it deliberately.
|
||||
- **A `.apm/` edit is not live in this session until it is pushed.** The six dependencies resolve from the holocron remote, unpinned against the default branch. `apm install` deploys from the lock; `apm update` is what re-resolves refs.
|
||||
- **The ADR-0020 skill gates ship hot, with no baseline.** 26 of 39 descriptions and 9 of 39 bodies exceed their FAIL tier, and the `Kyberforge.CompositionNote` Vale rule fires 10 errors across `gitea-issues`, `gitea-labels-milestones`, `gitea-prs` and `gitea-workflow`. Editing any of those skills *for any reason* means retrofitting it to the contract first — a one-line fix cannot be committed until the skill complies. Deliberate; tracked as Gitea issue #99. `skill-size-check` will not warn you about the Vale half, so check both: `pre-commit run --all-files`.
|
||||
- **Run `bash tests/run-tests.sh --strict` before considering any change done.** Keep the flag: without it a suite whose dependency is missing exits 77 and is counted SKIPPED rather than failed, so the run goes green having verified less than it claims.
|
||||
- **Before pushing, rehearse the gate locally:** `pre-commit run --hook-stage pre-push --all-files`. It runs the 14 pre-push hooks this repo authors itself plus pre-commit's 2 `meta` hooks, so it prints 16; `check-release-needed` passes without checking anything, because it needs a real push to `main`. `docs/spec/gates.md` reconciles both.
|
||||
- **Pushing without a network** needs `SKIP=apm-marketplace-check,apm-pack-check-clean git push` — those two resolve a remote marketplace entry via `git ls-remote`. Skip only those two; the rest are real local checks, and adding one to `SKIP` disarms it silently.
|
||||
- **Author commits with `git-commits`** — it validates Conventional Commits, which `commit-msg` enforces.
|
||||
- **This repo and Gitea are the only source of truth.** All project state, decisions, and working conventions live here. Do not use an external memory system for this project — cached state diverges from the repo and you get a split brain. Before answering any design or architecture question, check `docs/adr/` for an existing decision.
|
||||
|
||||
## Key documents
|
||||
|
||||
Read CONTEXT.md at the start of every session in this repo.
|
||||
Read `CONTEXT.md` at the start of every session — it is this repo's domain glossary, and the terms it defines are used unglossed everywhere else. It is not exhaustive: terms it does not carry are defined at their point of use, mostly in `docs/spec/`.
|
||||
|
||||
Read these on demand:
|
||||
|
||||
- `docs/spec/architecture.md` — current directory structure, install pipeline, provider model
|
||||
- `README.md` — prerequisites, install, and test commands
|
||||
- `docs/VISION.md` — the phased roadmap and where this is going; read when a decision turns on product direction
|
||||
- `LESSONS.md` — patterns that went wrong once; read before repeating a class of change that has burned the repo before
|
||||
- `docs/spec/gates.md` — what each pre-commit and pre-push hook enforces and why; read when a gate fails or before changing hook config
|
||||
- `docs/spec/architecture.md` — directory structure, install pipeline, provider model
|
||||
- `docs/adr/` — architectural decisions; read before answering design questions or proposing structural changes
|
||||
- `docs/ai-constitution.md` — full governance evidence base; read when a governance decision needs justification
|
||||
- `docs/research/ai-coding-factory/ai-coding-factory-principles.md` — factory design rationale; read when implementing, auditing, or reviewing skills or factory structure
|
||||
|
||||
246
CONTEXT.md
246
CONTEXT.md
@@ -1,80 +1,226 @@
|
||||
---
|
||||
name: AI Development Repo
|
||||
description: Domain language and decisions for the global AI development config repository
|
||||
description: The domain language of the global AI development config repository
|
||||
---
|
||||
|
||||
# Context
|
||||
# AI Development Repo
|
||||
|
||||
## Principles
|
||||
The bounded context of this repo is **how agent instructions are authored, packaged, distributed, and
|
||||
kept small**. Terms here name concepts specific to that problem. Mechanics live elsewhere:
|
||||
`docs/spec/architecture.md` for structure, `docs/spec/gates.md` for enforcement, `docs/adr/` for
|
||||
decisions.
|
||||
|
||||
### CLAUDE.md index model
|
||||
`AGENTS.md` is the source of always-on universal rules (provider-agnostic). `providers/claude-code/CLAUDE.md` is a thin adapter: it imports `~/.agents/AGENTS.md` via `@~/.agents/AGENTS.md` and appends Claude Code-specific additions (`@import` for governance.md, content index). Deployed to `~/.claude/CLAUDE.md` via `install.sh`. Context size is kept minimal — only what is needed every session is loaded upfront; detailed content is pulled on demand. See ADR-0003.
|
||||
## Language
|
||||
|
||||
### Instruction file format
|
||||
`core/instructions/<topic>.md` files are plain markdown — no frontmatter, no schema. The agent decides when to read each file based on task context and the content index label in `providers/claude-code/CLAUDE.md`. Frontmatter is deferred until there is evidence that agents are loading the wrong files in practice.
|
||||
### Context cost
|
||||
|
||||
### Repo/gitea as source of truth
|
||||
All project state, decisions, context, and working conventions live in this repo or Gitea. External memory systems should not be used for this project — they create a split-brain risk where cached state diverges from the repo. At the start of every session, read `CLAUDE.md`, `CONTEXT.md`, and `docs/VISION.md`. Everything needed to orient is here.
|
||||
**Preload tax**:
|
||||
The always-on context cost of every installed skill's `name` and `description`, charged from the
|
||||
first token of every session whether the skill is invoked or not. Measurement method and current
|
||||
figure: ADR-0020.
|
||||
_Avoid_: context cost, token overhead
|
||||
|
||||
Before answering any design or architecture question, check for existing decisions: `docs/adr/` (hard architectural decisions).
|
||||
**Skill context contract**:
|
||||
The ADR-0020 authoring rules that hold the preload tax and body size down — a description carries a
|
||||
trigger clause, at most one capability clause, and a boundary clause, and nothing else. Thresholds
|
||||
and the target-resolution walk: `docs/spec/gates.md`.
|
||||
_Avoid_: skill budget, size limit
|
||||
|
||||
## Glossary
|
||||
**Dispatch body**:
|
||||
The body pattern a skill with two or more mutually exclusive flows must use — the body carries only
|
||||
the dispatch table and the gates common to every branch, and each flow lives in its own
|
||||
self-contained `references/` file. Exemplar: `apm-workflow`.
|
||||
_Avoid_: router body, thin body
|
||||
|
||||
### Management Application
|
||||
A separate product (separate repo) for browsing, editing, and configuring AI development configs through a proper product UI. Git is the persistence layer, invisible to the user. The app is repo-agnostic — it works with any git repo that follows these conventions. This repo is the canonical default content (the official starter). See `docs/VISION.md` for the phased roadmap.
|
||||
**Hand-invoked skill**:
|
||||
A skill reached only by typing its slash command, declared `disable-model-invocation: true`. The host
|
||||
withholds it from the model-visible listing entirely, so it pays no preload tax and its description
|
||||
becomes human-facing text. Exemplar: `zoom-out`.
|
||||
_Avoid_: manual skill, disabled skill
|
||||
|
||||
### Skills
|
||||
Reusable slash commands for AI coding tools, defined as `SKILL.md` files following the [Agent Skills open standard](https://agentskills.io). Deployed via plugin — `plugins/<plugin-name>/.apm/skills/<skill-name>/SKILL.md`, available after the plugin is installed (`claude plugin install <name>@<marketplace>`). Skills are self-contained — they cannot reference files outside the plugin directory after install-time caching.
|
||||
**Delegation discipline**:
|
||||
The agent-side counterpart to the dispatch body. A plugin-scope agent is a single `.agent.md` file
|
||||
with no sibling `references/` directory, so it cannot disclose to itself — it can only delegate to
|
||||
skills. Its characteristic defect is therefore restatement, not length.
|
||||
_Avoid_: agent hygiene
|
||||
|
||||
### Plugin
|
||||
The deployable unit in the plugin marketplace. A plugin bundles one or more skills, agents, hooks, prompts, MCP servers, and optionally a `bin/` directory into a single installable directory. In this repo, plugins live under `plugins/<name>/`, each with its own `apm.yml` + `.apm/{skills,agents,hooks,...}` — this is the authoring source of truth for the plugin's content (ADR-0015). Two categories of tracked output are compiled from that source, never hand-edited: `.claude-plugin/plugin.json` (Claude Code) and `.github/plugin/plugin.json` (Copilot CLI) via `apm pack`/`apm compile`; and, alongside them, a flat `agents/`, `skills/`, `commands/`, `instructions/`, `extensions/` directory mirror at the plugin root plus a merged hooks file at `hooks/hooks.json`, generated by `scripts/sync-plugin-content.sh` — Claude Code's and Copilot's installers convention-scan only these flat paths (`hooks/hooks.json` is the convention path for hooks specifically; a root-level `hooks.json` is scanned by nothing and is deleted as stale by a sync — see ADR-0017's 2026-08-14 amendment) and have no awareness of `.apm/` nesting at all, so this mirror is what actually makes `.apm/` content discoverable at install time (ADR-0017). Plugins are copied to a cache on install — they cannot reference files outside their own directory. Install a plugin with `claude plugin install <name>@<marketplace>`.
|
||||
### Distribution
|
||||
|
||||
### Plugin marketplace
|
||||
A Git repository with a `marketplace.json` manifest listing installable plugins. No backend, registry, or SaaS required — the Git repo is the marketplace. This repo is the `holocron` marketplace. The manifest at `.claude-plugin/marketplace.json` (read by both Claude Code and Copilot CLI) is **compiled output** of `apm pack`, generated from the root `apm.yml`'s `marketplace:` block (owner, build/output config, versioning strategy, and the `packages:` list of installable plugins) — it is not hand-edited. See ADR-0015. `.github/plugin/marketplace.json` is Copilot CLI's legacy manifest path; apm has no output profile for it (only `claude` and `codex`, and `codex`'s is a differently-shaped file at `.agents/plugins/marketplace.json`), so `scripts/sync-marketplace-mirror.sh` keeps it byte-identical to `.claude-plugin/marketplace.json`, checked at pre-push. Each listed package's `source:` still points at that plugin's own `plugins/<name>/` root, not at an `apm pack` build artifact — which is why that root also carries the flat `agents/`/`skills/`/`commands/`/`hooks/hooks.json` content mirror described under "Plugin" (ADR-0017): without it, an install from this marketplace finds a valid manifest but no discoverable content.
|
||||
**Skill**:
|
||||
A reusable slash command defined as a `SKILL.md` file following the
|
||||
[Agent Skills open standard](https://agentskills.io), authored at
|
||||
`plugins/<plugin>/.apm/skills/<skill>/SKILL.md`.
|
||||
_Avoid_: command, prompt, macro
|
||||
|
||||
### HITL (human-in-the-loop)
|
||||
Agent pauses before a consequential action; human approves before execution. Required for irreversible or high-stakes actions (architecture changes, production deployments, security configuration). The agent drafts the change plan and waits — it does not proceed autonomously. Contrast with HOTL.
|
||||
**Plugin**:
|
||||
The deployable unit — one or more skills, agents, hooks, commands, and MCP servers bundled into a
|
||||
single installable directory under `plugins/<name>/`, compiled from that plugin's `.apm/` source.
|
||||
_Avoid_: package, bundle, module
|
||||
|
||||
### HOTL (human-on-the-loop)
|
||||
Agent acts; human monitors and can intervene after the fact. Acceptable for low-stakes, bounded, reversible actions where the cost of pausing for approval exceeds the blast radius of an error. The distinction between HITL and HOTL must be explicit and documented — defaulting to HOTL for convenience is not acceptable.
|
||||
**apm package**:
|
||||
The unit apm builds and installs — `plugins/<name>/apm.yml` plus the hand-authored
|
||||
`plugins/<name>/.apm/` tree it compiles from (ADR-0015).
|
||||
_Avoid_: plugin directory, source tree
|
||||
|
||||
### Sycophancy
|
||||
The failure mode where RLHF-trained models prioritise approval over accuracy. Treated as a first-class reliability risk: models change correct answers to wrong ones under user pressure in a majority of observed cases, then persist in the wrong answer. Designing against sycophancy is an explicit obligation, not a quality-of-life concern. Countermeasures: explicit pushback resistance instructions, prompting for dissent, cross-validating against independent sources. Never interpret AI agreement as AI accuracy.
|
||||
**Content mirror**:
|
||||
The generated flat `skills/`, `agents/`, `commands/`, `instructions/`, `extensions/` directories and
|
||||
merged `hooks/hooks.json` at a plugin root — also called the flat mirror — compiled from that
|
||||
plugin's `.apm/` tree so hosts that convention-scan those paths discover the content (ADR-0017).
|
||||
_Avoid_: generated copy, duplicate tree
|
||||
|
||||
### AGENTS.md
|
||||
The provider-agnostic always-on instruction entry point. Two files:
|
||||
- **Repo-level `AGENTS.md`** — instructions for agents working inside this repo (structure, key rules); imported by repo `CLAUDE.md` via `@AGENTS.md`.
|
||||
- **Global `core/AGENTS.md`** — Communication and Behavior rules that apply across all projects; deployed to `~/.agents/AGENTS.md`; imported by `~/.claude/CLAUDE.md` via `@~/.agents/AGENTS.md`.
|
||||
**Output profile**:
|
||||
An `apm pack` target format for a generated *marketplace* manifest; apm has `claude`
|
||||
(`.claude-plugin/marketplace.json`) and `codex` (the differently-shaped
|
||||
`.agents/plugins/marketplace.json`), and none for `.github/plugin/marketplace.json` (Copilot CLI's
|
||||
legacy path), which a sync script mirrors instead. Mechanics: `docs/spec/architecture.md`.
|
||||
_Avoid_: build target, export format
|
||||
|
||||
Contains always-on rules in plain markdown with no provider-specific syntax (no `@import`). Provider-specific files (`CLAUDE.md`) are thin adapters that import the relevant `AGENTS.md` and add only Claude Code-specific syntax. This pattern means a single source of truth can serve multiple providers without duplication. See ADR-0003.
|
||||
**Plugin marketplace**:
|
||||
A Git repository carrying a `marketplace.json` manifest that lists installable plugins. There is no
|
||||
backend, registry, or SaaS — the Git repo is the marketplace.
|
||||
_Avoid_: registry, store, catalogue
|
||||
|
||||
### Skill composition
|
||||
A skill calling another skill by name to delegate a sub-task. The calling skill focuses on the orchestration decision ("when to do X"); the called skill owns the mechanics ("how to do X"). Established compositions: `grill-me` calls `write-adr` when a decision crystallises; `implement-feature` calls `tdd` as its implementation methodology; `forge` calls `grill-with-docs` to refine intent, classifies the target artifact type (skill / agent / plugin / marketplace entry), then routes to the matching `*-author` skill — which owns its own create/improve logic and, where applicable, its own inline audit closeout (`skill-author` runs `/skill-audit`, `agent-author` runs `kyberforge:agent-audit`, both in the same context as the authoring work). Reserve `forge` for genuinely undecided "which artifact type is this" questions — an already-fully-specified corrective edit (exact file, line, and fix already known) should call the target author skill directly instead (`skill-author`, `apm-workflow`, `agentsmd-author`, etc.); routing a known fix through `forge`'s grill-and-classify layer adds unnecessary indirection and, in practice, has been observed to lose track of hard constraints handed down the chain (e.g. "don't commit yet," "edit in this worktree") because each hop re-derives instructions from a shorter brief. `forge` additionally runs its own independent recheck after a skill/agent route finishes: a clean-context subagent (not forked, no inherited context) re-runs the same audit skill against the finished artifact, as a distinct verification layer from the author skill's inline audit — the two can share blind spots since the inline audit runs in the same context as the work it checks. If the clean audit surfaces any unresolved finding, `forge` loops — re-invoke the author skill to resolve it, re-run the clean audit — until the clean audit comes back with nothing unresolved; only then is the route done. `plugin-author` and `marketplace-author` had no audit counterpart and got no recheck; their terminal check was `claude plugin validate`. Both were deprecated per ADR-0015, superseded by `apm-workflow`, and deleted entirely once issue #90 landed.
|
||||
**holocron**:
|
||||
This repository, in its role as a plugin marketplace and as the remote the six plugin dependencies
|
||||
resolve against.
|
||||
_Avoid_: the marketplace, upstream
|
||||
|
||||
### Provider-agnostic issue tracker
|
||||
Skills and workflows reference "linked issue" generically rather than a specific provider. Gitea is the canonical issue tracker for this repo (see ADR-0007). "Issue" is the cross-provider term (GitHub, GitLab, Gitea all use it).
|
||||
**apm-consumed install**:
|
||||
How this repo installs its own plugins as of 2026-08-14 — six `dependencies.apm` entries in the root
|
||||
`apm.yml` deployed by `apm install`, rather than `claude plugin install <name>@holocron`. Its
|
||||
consequences: ADR-0018.
|
||||
_Avoid_: apm install, dependency install
|
||||
|
||||
### Provenance chain
|
||||
The three-stage traceability record linking a skill back to its research inputs: (1) `/research` produces topic docs and a `sources.md` in `plugins/<plugin>/docs/research/docs/<topic>/`; (2) `/skill-author` reads those docs and records which sources informed which skill files in `references/sources.md` (including a `Research doc:` back-pointer to the upstream research file) and `source_keys` frontmatter on `SKILL.md` and `references/*.md`; (3) `skill-audit` validates the chain is complete and internally consistent via `validate-provenance.sh`. A skill with research input but no `references/sources.md`, or with `source_keys` that don't match `references/sources.md` slugs, has a broken provenance chain.
|
||||
**Provenance chain**:
|
||||
The three-stage traceability record linking a skill back to its research inputs: `/research` produces
|
||||
topic docs and a `sources.md`; the author skill records which sources informed which files in
|
||||
`references/sources.md` and `source_keys` frontmatter; `skill-audit` validates the chain is complete
|
||||
and internally consistent.
|
||||
_Avoid_: sources, citations, attribution
|
||||
|
||||
### Bidirectional reference principle
|
||||
Files that reference other files should declare those references explicitly. The referencing file carries the forward reference (e.g. content index in `CLAUDE.md`, `references:` in frontmatter). The referenced file carries a `when:` field describing when it is loaded. Both sides should agree — divergence signals staleness. The reverse map ("what files reference this file?") is derived by a reference scanner script, not maintained manually. This principle applies to instruction files, skills, and workflow documents.
|
||||
### Governance
|
||||
|
||||
### agentsmd-author / agentsmd-audit
|
||||
A skill pair in the `core` plugin for writing, updating, and reviewing a repo's `AGENTS.md` file(s) — the generic open-standard file (see the `AGENTS.md` entry above), including this repo's own. `agentsmd-author` creates/updates AGENTS.md content, supports nested monorepo placement (per the standard's nearest-file-wins precedence), and closes out by invoking `agentsmd-audit` inline. `agentsmd-audit` runs a single combined pass checking three mandatory baselines: secrets/credentials (governance.md hard prohibition — AGENTS.md is committed content), structural completeness (common-sections checklist from the agents.md spec), and accuracy/drift (do referenced commands and paths actually resolve against the repo). `agentsmd-audit` never inspects provider adapter files (see `provider-adapter-author`) — its scope is AGENTS.md content only. Chosen over folding this into `kyberforge` because kyberforge's scope is meta-tooling for the holocron marketplace itself, not generic target-repo documentation; `core` is the intended home for cross-cutting, repo-agnostic utility skills.
|
||||
**HITL** (human-in-the-loop):
|
||||
The agent pauses before a consequential action and a human approves before execution. Required for
|
||||
irreversible or high-stakes actions — architecture changes, production deployments, security
|
||||
configuration.
|
||||
_Avoid_: manual approval, gated action
|
||||
|
||||
### provider-adapter-author
|
||||
A companion skill (`core` plugin) that detects a target repo's provider-specific instruction file (`CLAUDE.md`, `.cursor/rules/*.mdc`, `copilot-instructions.md`, etc.) and, where it duplicates content AGENTS.md should own, converts it into a thin adapter that imports AGENTS.md — mirroring this repo's own ADR-0002/ADR-0003 two-tier adapter pattern. Self-validates via its own bundled deterministic script (`scripts/validate-adapter.sh`: checks for an import reference, no duplicated headings, size threshold) rather than a separate paired audit skill — the check is mechanical, so a script suffices per governance.md's "prefer deterministic code for repeatable tasks." `agentsmd-author` calls this skill via skill composition when it detects an existing provider file with overlapping content.
|
||||
**HOTL** (human-on-the-loop):
|
||||
The agent acts and a human monitors, able to intervene after the fact. Acceptable only for
|
||||
low-stakes, bounded, reversible actions where the cost of pausing exceeds the blast radius of an
|
||||
error.
|
||||
_Avoid_: autonomous, unsupervised
|
||||
|
||||
### lint plugin
|
||||
A standalone, repo-agnostic plugin (`plugins/lint/`) for configuring and running linters — not scoped to kyberforge's own meta-tooling. First linter is Vale (prose style linting), split into two skills per the git/gitea per-concern pattern: `vale-config` (setup — `.vale.ini`, `StylesPath`, styles) and `vale-run` (invoke Vale, interpret/report findings). A `lint-runner` agent composes these for isolated-context lint sweeps; it is report-only **by instruction, not by capability** — its body states "You never edit files" and "Do not edit, fix, or rewrite any flagged content", but nothing enforces that. It previously carried `tools: Bash, Read, Grep, Glob`, which withheld `Edit` outright; plugin-scope APM agents cannot express a `tools:` field at all (ADR-0016 — `apm compile` copies frontmatter verbatim to both Claude Code and Copilot, whose `tools:` vocabularies are incompatible, so a value correct for one harness is wrong for the other), so `plugins/lint/.apm/agents/lint-runner.agent.md` now declares only `name`/`description`/`source_keys` and inherits every tool, `Edit` included. ADR-0016 accepted this loss of enforcement knowingly; the restriction survives as prose the agent is expected to follow. Vale's research docs (`docs/research/docs/vale/`) moved from `plugins/kyberforge/` to `plugins/lint/` to keep the provenance chain same-plugin.
|
||||
**Sycophancy**:
|
||||
The failure mode where an RLHF-trained model prioritises approval over accuracy — changing a correct
|
||||
answer to a wrong one under user pressure, then persisting in the wrong answer. Treated here as a
|
||||
first-class reliability risk, not a quality-of-life concern.
|
||||
_Avoid_: agreeableness, people-pleasing
|
||||
|
||||
### Vale audit prefilter (skill-audit / agent-audit)
|
||||
Wiring Vale as a deterministic prefilter for `skill-audit`/`agent-audit`'s Description dimension (ADR motivation: issue #84) is repo-specific, not part of the generic `lint` plugin, so it doesn't live in `plugins/lint/` — but per ADR-0014 it also doesn't live at the repo root anymore. Two copies live inside `plugins/kyberforge/`, one per skill, since a plugin's cache-install only copies each skill's own files (no cross-skill sharing): `plugins/kyberforge/.apm/skills/agent-audit/assets/vale/` is canonical (`.vale.ini` plus a custom `Kyberforge` style covering description-opener banning ("This skill/agent..."), vague-capability wording ("helps with", "utilize", ...), and generic "see references/ for details" padding — and a `KyberforgeCopilot` style scoped only to `.agent.md` files for the Copilot-only "Use proactively has no effect" check), and `plugins/kyberforge/.apm/skills/skill-audit/assets/vale/` is a smaller duplicate (`Kyberforge` only, scoped to `SKILL.md`) kept in sync by `scripts/check-vale-style-sync.sh` (pre-push). A root-level `.pre-commit-hooks.yaml` exposes both copies (plus `skill-size-check`) so any external repo can enforce the same rules via `repo: <this-repo-url>, rev: <tag>` in its own `.pre-commit-config.yaml` — pre-commit clones the pinned rev into its own cache, independent of whether Claude Code or the `kyberforge` plugin is installed at all, and the same mechanism covers CI (`pre-commit run --all-files`). This repo's own `vale-audit-prefilter-skill`/`-agent` pre-commit hooks consume the identical plugin-bundled copies via `repo: local` (not a third root copy, and not a pinned self-reference — a pinned self-reference would lint working-tree edits against the last tagged release rather than the change being made). Every rule is `level: error` and every alert is a FAIL — no ignorable tier, same as shellcheck, the test suite, and conventional-pre-commit. Graded severities do not work here: Vale's exit code keys on `error` alerts alone, so `warning`/`suggestion` rules exit 0 and pre-commit swallows the output of a passing hook, leaving them invisible and blocking nothing. `MinAlertLevel` and `--minAlertLevel` are correspondingly absent from `.vale.ini` and the hook, being no-ops under this model. Vale covers the pattern-matchable sub-checks named in issue #84 (imperative opener, vague filler, `Use proactively`, generic reference-pointer padding) plus, per ADR-0013, one body-wide prose-pattern check ("There is/are" sentence openers) — everything else about body discipline (defaults-vs-menus, why-rationale, non-pattern-matchable judgment calls), near-miss exclusion strength, and control calibration stays LLM judgment.
|
||||
### Documents
|
||||
|
||||
Both skills' Step 1, and the `vale-audit-prefilter-skill`/`-agent` pre-commit hooks, call each copy's own `scripts/vale-wrap.sh` rather than `vale` directly — a workaround for a confirmed Vale 3.15.2 limitation (see `vale-config`'s Gotchas): `text.frontmatter.description` silently stops matching on most — not all — multi-line descriptions. Verified by reproduction, not assumed: `>` folded scalars, plain (unquoted) continuation lines, and single- or double-quoted multi-line scalars all yield 0 alerts and exit 0 on a deliberately-bad fixture, while a `|` literal block spanning the same 2+ lines lints normally (alerts fire, exit 1). The wrapper flattens those three broken forms to one physical line in a scratch copy (padding with blank lines so every other line number is unchanged) before handing off to real `vale`; `|` literal blocks and single-line descriptions pass through untouched, already linting correctly. The plain and quoted forms previously passed silently — unflattened and unmatched — so a bad description in either sailed through the prefilter. Handed no `--config` at all, the wrapper falls back to its own sibling `assets/vale/.vale.ini`, located from `${BASH_SOURCE[0]}` rather than from the cwd — which is why both manifests' `entry:` is now the bare script path with no argument after it. pre-commit prefixes only `entry[0]` with the hook-repo clone path (`cmd = (prefix.path(cmd[0]), *cmd[1:])`), so every later argument resolves against the *consuming* repo's root: a `--config` in `.pre-commit-hooks.yaml` pointed at a path no consumer has and hard-failed every external run with `E100 [--config] Runtime error`. `.pre-commit-config.yaml` drops the argument too, deliberately keeping the two entries identical — the local `repo: local` hook resolved its `--config` correctly only because the consuming repo *was* this repo, and that divergence is why three review rounds exercised a path no external consumer takes and missed the defect. An explicit `--config` still wins, in all three argv forms (`--config X`, `--config=/abs`, `--config=rel`), and a relative one still resolves against the caller's cwd, matching bare `vale`, not the repo root. Both audit skills' Step 1 now passes no `--config` either: it resolves the script relative to the skill's own directory so the call works from an installed plugin cache, but a relative `--config` alongside it would still resolve against the cwd, yielding `E100 Runtime error ... does not exist` and exit 2 — which both skills' fallback misreads as "vale unavailable" and silently downgrades to full LLM judgment. `tests/test-vale-wrap.sh` regression-tests this against skill-audit's copy specifically (its fixtures are all `SKILL.md`-shaped, and only skill-audit's `.vale.ini` has that glob section). Each `.vale.ini`'s section globs are path-agnostic (`[**/SKILL.md]` for skill-audit's copy; `[**/agents/*.md]`/`[**/*.agent.md]` for agent-audit's) and do no scoping on their own: Vale's `*` crosses `/`. Scoping comes from each pre-commit hook's own `files:` regex and from the audit skills passing one explicit file per invocation. The two manifests scope differently on purpose: this repo's `.pre-commit-config.yaml` pins its own layout — `^plugins/[^/]+/\.apm/skills/[^/]+/SKILL\.md$` for `-skill`, `^plugins/[^/]+/\.apm/agents/[^/]+\.agent\.md$` for `-agent` — while the shipped `.pre-commit-hooks.yaml` stays layout-agnostic for external consumers whose skills live anywhere, using `(^|/)SKILL\.md$` and `(^|/)agents/[^/]+\.md$|\.agent\.md$`. Both manifests split the prefilter into two hooks precisely because one combined hook pointed at only one copy would silently 0-file-skip the other file type. A `SKILL.md` outside `plugins/` (e.g. project-scope `.claude/skills/foo/SKILL.md`) still matches `[**/SKILL.md]` and gets linted normally — the globs constrain filename shape, not location. Vale reports 0 files only when the path it is handed matches no glob section at all: a differently-named file, or a directory argument holding nothing that matches. That run prints `✔ 0 errors ... in 0 files.` and exits 0, indistinguishable from a clean pass, so both audits treat a 0-file Vale run as NOT RUN and fall back to full LLM judgment.
|
||||
**AGENTS.md**:
|
||||
The provider-agnostic always-on instruction file, in plain markdown with no provider-specific syntax
|
||||
(ADR-0003). Two exist: repo-level, and the global `core/AGENTS.md` deployed to `~/.agents/AGENTS.md`.
|
||||
_Avoid_: instructions file, system prompt
|
||||
|
||||
This scope expands per ADR-0013: one cherry-picked low-noise `write-good`/`alex` rule landed in `styles/Kyberforge`, `Kyberforge.SentenceOpenerThereIs` (22 held-out hits, both in-corpus hits clean rewrites, zero suppressions). A second, `Kyberforge.VagueQualifier`, was cherry-picked and then deleted: 2 hits across the 41 skill/agent files, one marginal and one an unfixable false positive (`caveman/SKILL.md` quotes `of course` as an example of filler — a mention, not a use) that forced the repo's only Vale suppression comments. Also new is a sibling pre-commit hook, `skill-size-check` (`scripts/skill-size-check.sh`), enforcing agentskills.io's `SKILL.md` ceiling as two blocking gates: `MAX_LINES=500` and `MAX_WORDS=2770` (a word-count proxy for the 5,000-token limit, calibrated to the densest prose measured in this repo — 1.81 tokens per word — so even a worst-case `SKILL.md` at the ceiling stays under 5,000 tokens). Both are inclusive, and `skill-audit/scripts/validate.sh` checks the same pair on the same terms, so a `SKILL.md` can no longer pass its own audit yet be blocked by the commit hook. Scoped to `^plugins/[^/]+/\.apm/skills/[^/]+/SKILL\.md$` only, same as `vale-audit-prefilter-skill`, so it never lints `docs/research/examples/` reference skills. It's also exposed in the root-level `.pre-commit-hooks.yaml` as `kyberforge-skill-size-check` — it has no external asset dependency, so it needed no relocation, only exposure to external consumers. File scope (`SKILL.md` + agent files) and enforcement model (rules land directly in `styles/Kyberforge`, blocking immediately, no trial tier) stay unchanged; governance.md/CONTROLS.md were evaluated and excluded as rule sources (nothing prose-pattern-matchable to mine). House convention: banned phrasing that must be mentioned rather than used goes in backticks or a fenced code block — Vale skips code spans and fences, so no suppression is needed; inline `<!-- vale Rule = NO -->` (HTML-comment form; the MDX `{/* */}` form does not work in plain Markdown) is the fallback only where backticking is impossible.
|
||||
**Thin adapter**:
|
||||
A provider-specific instruction file (`CLAUDE.md`, `.cursor/rules/*.mdc`, `copilot-instructions.md`)
|
||||
that imports its `AGENTS.md` and adds only that provider's syntax, carrying no original always-on
|
||||
content of its own (ADR-0002, ADR-0003).
|
||||
_Avoid_: wrapper, shim, provider file
|
||||
|
||||
### LESSONS.md
|
||||
Long-loop feedback log for patterns observed across sessions. Three or more entries on the same pattern graduate to the relevant standing file (e.g. a coding convention, a governance rule). Updated by the session-handoff skill or directly by the human. Lives at the repo root.
|
||||
**LESSONS.md**:
|
||||
The long-loop feedback log for patterns observed across sessions, at the repo root.
|
||||
_Avoid_: changelog, retro, postmortem
|
||||
|
||||
**Management Application**:
|
||||
A separate product in a separate repo for browsing, editing, and configuring AI development configs
|
||||
through a product UI, with Git as an invisible persistence layer. Repo-agnostic; this repo is its
|
||||
canonical default content. Roadmap: `docs/VISION.md`.
|
||||
_Avoid_: the UI, the dashboard, the app
|
||||
|
||||
### Quality
|
||||
|
||||
**Skill composition**:
|
||||
A skill calling another skill by name to delegate a sub-task — the caller owns the orchestration
|
||||
decision ("when to do X"), the callee owns the mechanics ("how to do X").
|
||||
_Avoid_: chaining, nesting, sub-skill
|
||||
|
||||
**Vale audit prefilter**:
|
||||
The deterministic Vale pass that runs ahead of `skill-audit`/`agent-audit`'s Description dimension,
|
||||
so LLM judgment is spent only on what a pattern cannot catch. Mechanics: `docs/spec/gates.md`.
|
||||
_Avoid_: linting, style check
|
||||
|
||||
**Authoring root**:
|
||||
The directory a gate resolves against — the nearest ancestor of the file being checked holding
|
||||
`plugins/*/.apm/skills` or `plugins/*/.apm/agents`, falling back to the nearest ancestor holding
|
||||
`.git`. The walk: `docs/spec/gates.md`.
|
||||
_Avoid_: repo root, project root
|
||||
|
||||
**Near-miss**:
|
||||
A query that shares keywords with this skill but needs a different one — and, by extension, the
|
||||
sibling that would wrongly answer it; boundary clauses exist to exclude genuine near-misses rather
|
||||
than to enumerate siblings. Detail: `skill-audit/references/description-quality.md`.
|
||||
_Avoid_: overlap, similar skill
|
||||
|
||||
**Vacuous green**:
|
||||
A check that reports success because it measured nothing — zero files scanned, an unparsed value read
|
||||
as empty, a conditional branch that never armed.
|
||||
_Avoid_: false pass, clean run
|
||||
|
||||
**Issue**:
|
||||
The cross-provider term for a tracked unit of work. Gitea is this repo's canonical tracker
|
||||
(ADR-0007), but skills say "linked issue" generically rather than naming a provider.
|
||||
_Avoid_: ticket, card, task
|
||||
|
||||
## Relationships
|
||||
|
||||
- A **Plugin** bundles one or more **Skills** and agents; a **Plugin marketplace** lists **Plugins**;
|
||||
**holocron** is this repo wearing that hat.
|
||||
- Every model-invocable **Skill** pays the **Preload tax**. A **Hand-invoked skill** does not — which
|
||||
is the first question to settle when authoring one.
|
||||
- The **Skill context contract** bounds both the **Preload tax** (description) and the body.
|
||||
A **Dispatch body** is how a skill stays inside it; **Delegation discipline** is how an agent does.
|
||||
- **AGENTS.md** is the source of always-on rules; a **Thin adapter** imports it and originates
|
||||
nothing.
|
||||
- **Skill composition** is the caller/callee split. `forge` routes a genuinely *undecided* artifact
|
||||
type to the matching author skill — an already-specified fix (file, line, and change known) calls
|
||||
that author skill directly, because each routing hop re-derives instructions from a shorter brief
|
||||
and has been observed to drop hard constraints handed down the chain.
|
||||
- **HITL** and **HOTL** are exclusive per action class, and the choice must be explicit and
|
||||
documented. **Sycophancy** is why HOTL is not the safe default.
|
||||
- A **Skill** built on research carries a **Provenance chain**; `skill-audit` fails it when broken.
|
||||
- **LESSONS.md** feeds the standing files: three or more entries on one pattern graduate the pattern
|
||||
into the relevant standing document.
|
||||
|
||||
## Example dialogue
|
||||
|
||||
> **Dev:** "This one only fires when someone types the slash command. Does its description still need
|
||||
> trigger words?"
|
||||
> **Maintainer:** "No — that's a **hand-invoked skill**. The host withholds it from the model-visible
|
||||
> listing, so it pays no **preload tax** at all and the description is human-facing text."
|
||||
> **Dev:** "Then the body can be as long as it needs to be?"
|
||||
> **Maintainer:** "Different budget. The **skill context contract** gates the body whether or not the
|
||||
> skill is model-invoked — the description competes with every other skill's description, the body
|
||||
> competes with the caller's live conversation. Four mutually exclusive flows means a **dispatch
|
||||
> body**: table in `SKILL.md`, one `references/` file per flow."
|
||||
> **Dev:** "And if I split it into an agent instead?"
|
||||
> **Maintainer:** "Then you're in **delegation discipline** territory. An agent has no `references/`
|
||||
> to disclose to, so the failure mode flips — it stops being length and starts being restatement of
|
||||
> a procedure some skill already owns."
|
||||
|
||||
## Flagged ambiguities
|
||||
|
||||
- "skill" was used for both the authored `SKILL.md` under `plugins/<name>/.apm/skills/` and the
|
||||
deployed copy under `.claude/skills/` — resolved: the authoring source is the **Skill**; the
|
||||
deployed copy is gitignored `apm install` output and is never edited.
|
||||
- Skills can answer to two names, bare (`gitea-prs`) and namespaced (`gitea:gitea-prs`), depending on
|
||||
whether a native install exists at user scope alongside the apm one (ADR-0018) — resolved: write
|
||||
the bare name, which is the only form `apm install` produces.
|
||||
- "context" means both the model's live token window (the **Preload tax** sense) and the bounded
|
||||
domain this file describes — resolved: unqualified "context" in this repo means the token window.
|
||||
- "audit" was used for both an author skill's inline closeout and `forge`'s independent
|
||||
clean-context recheck — resolved: these are two distinct layers, kept separate precisely because
|
||||
an audit running in the same context as the work it checks shares that work's blind spots.
|
||||
|
||||
90
LESSONS.md
90
LESSONS.md
@@ -1,8 +1,8 @@
|
||||
# Lessons
|
||||
|
||||
Patterns observed during development of this repo. Three or more entries on the same pattern → promote to CONTEXT.md (or the relevant instruction file) as a standing rule.
|
||||
Patterns observed during development of this repo. Three or more entries on the same pattern → promote to `docs/spec/architecture.md` (or the relevant instruction file) as a standing rule.
|
||||
|
||||
**Graduation rule:** When three or more entries cover the same pattern, the human reviews and promotes it to the appropriate standing location: `CONTEXT.md` for domain-level principles, `core/instructions/coding.md` for coding conventions, `core/instructions/git.md` for git conventions, or `core/instructions/testing.md` for testing conventions. The graduated entries are marked `[graduated → target file]` rather than deleted (audit trail).
|
||||
**Graduation rule:** When three or more entries cover the same pattern, the human reviews and promotes it to the appropriate standing location: `docs/spec/architecture.md` for structural and domain-level principles — `CONTEXT.md` is not a destination, its `## Principles` section was deleted and what was there now sits under that file's "AGENTS.md pattern" and "Reference conventions" headings — `core/instructions/coding.md` for coding conventions, `core/instructions/testing.md` for testing conventions, or `core/instructions/subagent-orchestration.md` for delegation conventions. Those four are the whole set — `core/instructions/` holds `coding.md`, `governance.md`, `subagent-orchestration.md` and `testing.md`, and nothing else. Git conventions have no standing file of their own: promote them to `core/instructions/coding.md`, or create a new instruction file deliberately rather than assuming one exists. The graduated entries are marked `[graduated → target file]` rather than deleted (audit trail).
|
||||
|
||||
**Who writes here:** The session-handoff skill (Chunk 3) prompts LESSONS.md extraction before closing a session. The human may also write directly.
|
||||
|
||||
@@ -26,6 +26,8 @@ Issue files frequently referenced "the workflow defined in `docs/notes/skill-imp
|
||||
|
||||
The repo CLAUDE.md instructs agents to read CONTEXT.md at session start, but agents skip this in practice — defaulting to reading only what's directly relevant to the immediate prompt (e.g. the skills folder). The governance.md works because `@import` is technically enforced by Claude Code. Fix: (1) add `@CONTEXT.md` to repo CLAUDE.md using `@import` to make it always-loaded; (2) add a "Key decisions" section to CONTEXT.md with one-line resolved-ADR summaries so locked choices are always in context.
|
||||
|
||||
**Status (2026-08-14): neither part landed.** Root `CLAUDE.md` imports `@AGENTS.md` only — no `@CONTEXT.md` — and `CONTEXT.md` has no "Key decisions" section. The behavioral hope this entry diagnosed is still the only mechanism in place: `AGENTS.md` carries the line "Read `CONTEXT.md` at the start of every session," which is loaded but is itself an instruction, not an import. The proposal above is open work, not a record of a completed change.
|
||||
|
||||
## 2026-05-17 — Instruction rules lose to RLHF defaults without specificity
|
||||
|
||||
Behavioral tests (2026-05-17) showed three communication/behavior rules failing: exploratory question format (gave verbose multi-bullet answer instead of 2-3 sentences), file edit intent (asked for clarification instead of stating intent and proceeding), and push confirmation (went straight to tool call instead of asking first). All three rules are present in `providers/claude-code/CLAUDE.md` as one-liner statements. The RLHF-trained defaults (thorough answers, risk-averse clarification seeking, fast execution) consistently outcompete thin rules. Fix: rewrite failing rules with specificity, a counter-example, and a boundary statement — not just a single-line imperative.
|
||||
@@ -175,3 +177,87 @@ Across one review round, four fixes specified by the orchestrating reviewer were
|
||||
Mutation testing a review round's own fixes found repeatedly that a passing test was pinning nothing. Deleting `sync_dir`'s stale-directory wipe, its check-mode stale branch, or three of five `MIRROR_DIRS` entries each left the suite at 18/18 green; so did replacing the hooks trailing-newline normalisation with plain `cp`. A pair of concurrency assertions written to guard a reentrancy defect caught it 0 times in 10 runs against the deliberately broken script — and one of them was structurally incapable of ever catching it, because the broken code wrote to the system temp dir while the assertion inspected `$TMPDIR`. A fixture-leak fix ran green with and without the fix, verified only by external observation. Two manifest fixtures passed with the canonicalisation they claimed to cover deleted, rescued by an unrelated name-matching axis. In each case the test named the right behaviour in its description and asserted something adjacent to it. The cheap discipline that finds all of these: for every assertion, construct the revert it is supposed to catch and confirm it fails — and when an assertion survives every revert you can think of, that is not reassurance, it is the finding (one test only revealed itself as decoration once a sixth, differently-targeted revert was built for it). Fix: treat "which revert does this fail against?" as a required answer at the time an assertion is written, and record it where the assertion lives, since a test's own description is exactly the artifact that made the gap invisible.
|
||||
|
||||
Graduation candidate: this overlaps 2026-08-09's "an assertion written to cover an accepted residual tends to assert the residual's presence rather than the behaviour it costs" and the same date's "assert on the expected members, so a derivation whose input vanished fails loudly instead of quietly covering less." Three entries circling one pattern — human review for promotion to `core/instructions/testing.md`.
|
||||
|
||||
## 2026-08-14 — Vale's `existence` extension concatenates `raw:` entries, it does not alternate them
|
||||
|
||||
A new `Kyberforge.CompositionNote` rule was first written with seven `raw:` entries, one per banned
|
||||
phrasing. Vale loaded it without a diagnostic and it matched **zero of 43 files** — an outcome
|
||||
indistinguishable from a clean corpus, and the exact shape of 2026-08-08's "a clean linter result can
|
||||
mean nothing was checked". The cause is that `existence` joins multiple `raw:` entries into one
|
||||
pattern rather than OR-ing them, so the rule was searching for all seven phrases concatenated. Every
|
||||
pre-existing rule in this style has exactly one `raw:` entry, so nothing in the repo demonstrated the
|
||||
difference, and the multi-entry form looks natural beside them. `tokens:` is the alternated form,
|
||||
which is why `VagueWording` uses it. Fix: a new Vale rule is not landed until it has been shown to
|
||||
*fire* — the standing revert-check applies to linter rules as much as to tests, and the revert here
|
||||
is the broken multi-`raw:` form, which `tests/test-vale-hooks-consumer.sh` now fails against.
|
||||
|
||||
## 2026-08-14 — Un-anchoring a description rule to reach mid-sentence text is unshippable
|
||||
|
||||
Widening `DescriptionOpener` to catch `gitea-workflow`'s mid-description "This is the human-facing
|
||||
entry point…" looked like a one-character change. Both that skill and `gitea-labels-milestones`
|
||||
*open* with "Use when…" and satisfy the opener rule; the offending clause sits at character 377 and
|
||||
300 of the folded value respectively, so the rule was never violated and never silently passed — it
|
||||
simply had no jurisdiction, which is a different defect and takes a different fix.
|
||||
Under `scope: text.frontmatter.description`, `^`
|
||||
anchors to the start of the whole description value — and `vale-wrap.sh` has already flattened that
|
||||
value to one physical line, so `(?m)` changes nothing. Un-anchoring is therefore the only route to
|
||||
mid-description text, and measured across the corpus it scores 5 hits and 5 false positives: skills
|
||||
legitimately quote user phrasings (`says "audit this skill"`) and write boundary clauses (`do not use
|
||||
this skill to manage label definitions`). That is the `Kyberforge.VagueQualifier` deletion repeating.
|
||||
Fix: keep the opener rule opener-anchored and give mid-description prose its own rule with its own
|
||||
token list. A rule's scope anchor is part of its contract, not an implementation detail to relax when
|
||||
a new case does not fit.
|
||||
|
||||
## 2026-08-14 — A formatter in the commit path manufactures drift on a file with a clean git diff
|
||||
|
||||
`apm audit --ci` failed on `.claude/settings.json` while `git diff` on that file was empty — the worst
|
||||
possible pairing of signals, because the file matched HEAD exactly and every instinct says "nothing
|
||||
changed here". The content was identical to apm's output to the byte; only the JSON key order
|
||||
differed. `pretty-format-json --autofix` sorts object keys unless `--no-sort-keys` is passed, and its
|
||||
`exclude:` listed fifteen generated manifests but not this file, so from the commit that first wrote
|
||||
a hook entry there onward, apm's insertion-ordered output was silently re-sorted on the way in. apm
|
||||
then replayed the install, produced its own order, and reported drift against a file no human had
|
||||
touched.
|
||||
|
||||
The provenance matters as much as the mechanism, and the first account of this entry got it wrong in
|
||||
both directions. `git log --format='%h %ad %s' --date=iso` puts the introducing commit `2e395a4` at
|
||||
2026-08-14 18:47 and the fix `7607522` at 21:54 — roughly three hours, not "weeks". And `2e395a4` is
|
||||
the **first commit of the `refactor/trim-skills-agents-context` branch**, eleven minutes after the
|
||||
base merge `f9b919d`; `git branch -a --contains 2e395a4` returns only that branch and its own
|
||||
`remotes/origin/` tracking copy — two lines naming one branch, and `main` is not among them. So
|
||||
this was not a latent defect inherited from `main`, it was manufactured inside the same PR that
|
||||
diagnosed it, and the fixing commit's own message calling it "pre-existing … red at HEAD before
|
||||
ADR-0020 work began" is the mis-attribution rather than the record. Two cheap commands would have
|
||||
settled it before either sentence was written.
|
||||
|
||||
Three general points. First, a tool-owned generated file that passes through an autofixing formatter
|
||||
is drifted by construction, and the diff that would reveal it never appears in `git diff` — it only
|
||||
exists between the formatter's input and its output, which nothing stores. Second, the fix is
|
||||
self-undoing unless the exclude lands in the same commit: correcting the file alone means the hook
|
||||
re-breaks it as it is staged. Third — the one this entry had to learn twice — "pre-existing" is a
|
||||
claim about history, and history is queryable; a defect found while working on a branch feels
|
||||
inherited, and the feeling is not evidence. A three-hour-old self-inflicted bug and a months-old
|
||||
inherited one call for different responses, and writing the wrong one down converts a process failure
|
||||
into a story about someone else's neglect. Fix: when a tool declares ownership of a path, add that
|
||||
path to every autofixing hook's `exclude` at the moment ownership is declared, not when the drift is
|
||||
noticed — and before describing any defect as pre-existing, run `git log -S` or
|
||||
`git branch --contains` on the commit that introduced it. This repo gates marketplace-mirror,
|
||||
plugin-content and vale-style drift deterministically and has no equivalent gate asserting tool-owned
|
||||
paths stay out of formatter scope — `.claude/settings.json` was the sixteenth exclude and nothing
|
||||
prevents a seventeenth.
|
||||
|
||||
## 2026-08-16 — A rule reversed inside a retrofit leaves no trace unless someone writes it down
|
||||
|
||||
`skill-author/SKILL.md:204` on `main` said "Keep reference chains one level deep — a reference file
|
||||
that references another reference file is rarely loaded correctly." The ADR-0020 retrofit replaced it
|
||||
with "Two hops from `SKILL.md`, never three" in `references/create.md` and `references/retrofit.md`,
|
||||
which permits exactly the chain the old rule banned. The looser rule is the right one and the
|
||||
retrofit could not have shipped without it: dispatch pushes each flow into its own file, so the
|
||||
shipped structure is `SKILL.md` → `improve.md` → `retrofit.md`, and a one-level ceiling would have
|
||||
made the mandatory dispatch pattern illegal. But ADR-0020 says nothing about chain depth, so the
|
||||
reversal was carried entirely by the diff — the new text asserts the new rule with no sign that a
|
||||
contradicting rule ever existed, and a reader who remembers the old one has no way to tell whether it
|
||||
was overturned or overlooked. Fix: when a change inverts a standing authoring rule rather than
|
||||
tightening or restating it, record the inversion where the rule's rationale lives — the ADR if the
|
||||
ADR is the reason, here otherwise. A rule that quietly flips is indistinguishable from a rule that
|
||||
was forgotten, and the second reading is the one that gets it re-added later.
|
||||
|
||||
131
README.md
Normal file
131
README.md
Normal file
@@ -0,0 +1,131 @@
|
||||
# holocron
|
||||
|
||||
The global AI development configuration repository — the authoritative source for agent definitions, skills, workflows, and prompts across all projects. Built as a homelab tool intended to scale to professional environments.
|
||||
|
||||
Content ships as six installable plugins, each an apm (Agent Package Manager) package. This repo consumes its own plugins through apm, so the working copy runs the same released content every other consumer gets.
|
||||
|
||||
## Repo layout
|
||||
|
||||
| Path | What it holds |
|
||||
| --- | --- |
|
||||
| `plugins/` | Six apm packages — `bin`, `core`, `git`, `gitea`, `kyberforge`, `lint` — each carrying skills, and where relevant agents, hooks, MCP servers, and bundled assets |
|
||||
| `providers/claude-code/` | Claude Code adapter, deployed to `~/.claude/` via `scripts/install.sh` |
|
||||
| `core/` | Provider-agnostic always-on content — `core/AGENTS.md` and `core/instructions/` |
|
||||
| `docs/` | Specs (`docs/spec/`), architectural decisions (`docs/adr/`), governance, research, and notes |
|
||||
| `scripts/` | Install, sync, and check scripts used by the git hooks |
|
||||
| `tests/` | `run-tests.sh`, `run-bats.sh`, the `test-*.sh` suites, and the bats submodules |
|
||||
|
||||
The six plugins:
|
||||
|
||||
- **kyberforge** — skills and agents for creating, maintaining, and managing a Claude Code / Copilot CLI plugin marketplace
|
||||
- **git** — conventional commits, branches, history, submodules, worktrees, remotes, pre-commit hook authoring and running (`pc-author` / `pc-run`), and an interactive router (`git-workflow`)
|
||||
- **gitea** — issues, pull requests, labels, milestones, releases, branches, files, and an interactive router (`gitea-workflow`)
|
||||
- **core** — authoring and auditing a repo's `AGENTS.md` and the provider adapter files that defer to it
|
||||
- **lint** — configuring and running linters
|
||||
- **bin** — cross-cutting workflow skills not yet split into a focused plugin: research, documentation, TDD, prototyping, triage, diagnosis, architecture review, requirement grilling, compressed output (`caveman`), and re-orienting mid-task (`zoom-out`)
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Install all of these before setting up. Each one is a hard dependency of a git hook or a script — several fail with an unhelpful "command not found" if missing.
|
||||
|
||||
| Tool | Why | Install |
|
||||
| --- | --- | --- |
|
||||
| `apm` CLI | Four pre-push hooks shell out to it (`apm-marketplace-check`, `apm-audit-ci`, `apm-pack-check-clean`, and `check-plugin-content-sync` via `scripts/sync-plugin-content.sh`) | The `apm-install` skill, or `curl -sSL https://aka.ms/apm-unix \| sh`. Verify with `apm --version` |
|
||||
| `jq` | Required by `scripts/check-manifests.sh` and `scripts/sync-plugin-content.sh`, both pre-push | Your package manager |
|
||||
| `python3` + PyYAML | Required by `scripts/skill-size-check.sh` (the `skill-size-check` pre-commit hook), which reads folded YAML frontmatter | `python3` is usually present — pre-commit is itself a Python application. `pip install pyyaml` if the hook reports PyYAML missing |
|
||||
| `vale` | Required by the `vale-audit-prefilter-skill` / `-agent` pre-commit hooks and the `check-vale-style-sync` pre-push hook | `brew install vale` (macOS), `snap install vale` (Linux), `choco install vale` (Windows), or https://vale.sh/docs/vale-cli/installation/ |
|
||||
| `claude` CLI | Required by the `validate-plugins` and `validate-marketplace` pre-push hooks | Claude Code |
|
||||
|
||||
Two notes worth reading before you skip one:
|
||||
|
||||
- **PyYAML is a hard requirement, not an optional accelerator.** The hand-rolled fallback frontmatter reader was removed deliberately: a reader that mis-parses an unfamiliar scalar shape reports a clean pass on a file it never measured.
|
||||
- **No `vale sync` is needed.** The `Kyberforge` styles are committed under `plugins/kyberforge/.apm/skills/{skill-audit,agent-audit}/assets/vale/styles/`, not downloaded packages (ADR-0014).
|
||||
|
||||
## Setup
|
||||
|
||||
Run these in order, from the repo root.
|
||||
|
||||
```bash
|
||||
# 1. Deploy this repo's own skills and agents
|
||||
apm install
|
||||
|
||||
# 2. Install the git hooks — all three stages
|
||||
pre-commit install -t pre-commit -t commit-msg -t pre-push
|
||||
```
|
||||
|
||||
**`apm install`** deploys the six plugins into `.claude/skills/` and `.claude/agents/`. Both are gitignored install output, *not* authoring source — `plugins/<name>/.apm/` remains the only place to edit. It needs the network, materializes `apm_modules/` (which stays gitignored), and also configures the `obsidian` MCP server into the repo's `.mcp.json`.
|
||||
|
||||
**Git hooks** must be wired for **all three stages**. This repo's `.pre-commit-config.yaml` has no `default_install_hook_types`, so a plain `pre-commit install` silently skips `commit-msg` (Conventional Commits) and `pre-push` (the full gate) — the `-t` flags above are not optional. The `pc-run` skill handles this and the troubleshooting around it, if you would rather not remember the flags.
|
||||
|
||||
## Keeping the install current
|
||||
|
||||
The six dependencies in root `apm.yml` are unpinned against the default branch, so deployed skills go stale whenever anyone merges. kyberforge ships a `SessionStart` hook that runs `apm outdated` at startup (~0.7s) and, when something is behind, runs `apm update --yes` and asks the host to re-scan skills (~10.4s).
|
||||
|
||||
That rewrites `apm.lock.yaml` — an unexplained modification to it after opening a session is expected, not a bug. Commit or discard it deliberately.
|
||||
|
||||
Note the difference between the two commands:
|
||||
|
||||
- `apm install` deploys from `apm.lock.yaml`. It does **not** pick up remote changes.
|
||||
- `apm update` re-resolves refs. This is the command that pulls in a merged `.apm/` edit.
|
||||
|
||||
## Running tests
|
||||
|
||||
```bash
|
||||
bash tests/run-tests.sh # every test-*.sh script plus the bats suite
|
||||
bash tests/run-tests.sh --bats-only # just bats
|
||||
```
|
||||
|
||||
The first run auto-initializes the bats submodules; no manual `git submodule update` needed.
|
||||
|
||||
A suite that exits 77 because a dependency is missing is reported as SKIPPED and does **not** fail an ad-hoc run. It *does* fail under `--strict` (equivalently `RUN_TESTS_STRICT=1`), which is how the pre-push hook invokes it — at pre-push, a skip means one of the prerequisites above is absent on this machine, and the gate would otherwise report success having run fewer suites than it appears to. The strict failure names each skipped suite and what to install.
|
||||
|
||||
## Before pushing
|
||||
|
||||
Run the pre-push gate locally in one command:
|
||||
|
||||
```bash
|
||||
pre-commit run --hook-stage pre-push --all-files
|
||||
```
|
||||
|
||||
One caveat: `check-release-needed` is a silent no-op under this invocation. It exits 0 unless
|
||||
`PRE_COMMIT_REMOTE_BRANCH` is `refs/heads/main`, and pre-commit exports that only from the real
|
||||
pre-push git hook during an actual `git push` — so the hook reports `Passed` having checked nothing.
|
||||
Every other pre-push hook does run.
|
||||
|
||||
See [`docs/spec/gates.md`](docs/spec/gates.md) for what each hook enforces and why.
|
||||
|
||||
**Offline?** Exactly two pre-push hooks need the network, because root `apm.yml`'s marketplace contains one remote package entry that must be resolved with `git ls-remote`:
|
||||
|
||||
```bash
|
||||
SKIP=apm-marketplace-check,apm-pack-check-clean git push
|
||||
```
|
||||
|
||||
Skip **only** those two. The remaining pre-push hooks are real local checks and pass offline; adding one of them to `SKIP` disarms it silently.
|
||||
|
||||
## Editing plugin content
|
||||
|
||||
`plugins/<name>/.apm/` is the only hand-edited source for plugin content — skills, agents, commands, instructions, extensions, and hooks. The flat `plugins/<name>/{skills,agents,commands,instructions,extensions}/` directories, the merged `hooks/hooks.json`, and every `plugin.json` / `marketplace.json` manifest are generated. Nothing labels a generated file as generated, so check the path before you edit; an edit to the mirror is discarded by the next sync and reported as drift by the `check-plugin-content-sync` pre-push hook.
|
||||
|
||||
Hand-authored material that is *not* an `.apm/` primitive — `README.md`, `docs/`, `bin/`, `sources.md`, `.mcp.json` — lives at the plugin **root** and is untouched. Never place such a file inside a mirrored directory: the sync removes the destination before every copy, so it is deleted with no drift report.
|
||||
|
||||
Full detail in [`docs/spec/architecture.md`](docs/spec/architecture.md).
|
||||
|
||||
## For external consumers
|
||||
|
||||
Install a plugin natively from the marketplace manifests:
|
||||
|
||||
```bash
|
||||
claude plugin install <name>@holocron
|
||||
```
|
||||
|
||||
Or consume the packages through apm, the way this repo does — declare them as `dependencies.apm` git+path entries against the holocron remote and run `apm install`.
|
||||
|
||||
## Where to go next
|
||||
|
||||
- [`AGENTS.md`](AGENTS.md) — the rules for AI agents working in this repo
|
||||
- [`CONTEXT.md`](CONTEXT.md) — domain language; read at the start of every session here
|
||||
- [`docs/spec/architecture.md`](docs/spec/architecture.md) — directory structure, install pipeline, provider model
|
||||
- [`docs/spec/gates.md`](docs/spec/gates.md) — the enforcement gates in depth
|
||||
- [`docs/adr/`](docs/adr/) — architectural decisions; read before proposing structural changes
|
||||
- [`docs/VISION.md`](docs/VISION.md) — where this is going
|
||||
- [`LESSONS.md`](LESSONS.md) — things that went wrong once and should not again
|
||||
2936
apm.lock.yaml
Normal file
2936
apm.lock.yaml
Normal file
File diff suppressed because it is too large
Load Diff
67
apm.yml
67
apm.yml
@@ -1,16 +1,61 @@
|
||||
name: holocron
|
||||
version: 0.3.3
|
||||
version: 0.4.5
|
||||
description: AI development skills for Claude Code and GitHub Copilot CLI — factory, design, implement, review, and cross-cutting workflows.
|
||||
license: MIT
|
||||
|
||||
# Consumer side: this repo installs its own published plugins from the holocron
|
||||
# remote, so the working copy runs the same released content every other
|
||||
# consumer gets. Addressed as git+path objects rather than <name>@holocron
|
||||
# marketplace aliases — an alias needs a `apm marketplace add` registration in
|
||||
# ~/.apm/marketplaces.json (user scope, outside this repo), the object form
|
||||
# needs nothing beyond this manifest.
|
||||
# Unpinned (default branch) on purpose: parity with the Claude Code plugin
|
||||
# install this replaced, which ran autoUpdate against main. Add `ref: <tag>`
|
||||
# per entry to pin.
|
||||
targets:
|
||||
- claude
|
||||
dependencies:
|
||||
apm:
|
||||
- git: git@git.dev.rkdr.net:Defame1297/holocron.git
|
||||
path: plugins/bin
|
||||
- git: git@git.dev.rkdr.net:Defame1297/holocron.git
|
||||
path: plugins/core
|
||||
- git: git@git.dev.rkdr.net:Defame1297/holocron.git
|
||||
path: plugins/git
|
||||
- git: git@git.dev.rkdr.net:Defame1297/holocron.git
|
||||
path: plugins/gitea
|
||||
- git: git@git.dev.rkdr.net:Defame1297/holocron.git
|
||||
path: plugins/kyberforge
|
||||
- git: git@git.dev.rkdr.net:Defame1297/holocron.git
|
||||
path: plugins/lint
|
||||
mcp: []
|
||||
|
||||
# Turns apm's executable-trust gate ON. Without this block the gate is disabled
|
||||
# and every hook, bin and MCP primitive a dependency ships deploys silently —
|
||||
# verified: `apm approve --list` reports "Executable-trust gate disabled -- all
|
||||
# executables deploy" until an `executables:` block exists.
|
||||
#
|
||||
# kyberforge ships the SessionStart hook that keeps this install level with the
|
||||
# remote (ADR-0019). The key is version-pinned by apm's own design, so a
|
||||
# kyberforge version bump makes this entry stop matching and the hook stops
|
||||
# deploying until the version here is bumped too. If skills silently go stale
|
||||
# after a kyberforge release, check this first.
|
||||
executables:
|
||||
allow:
|
||||
kyberforge#1.6.0:
|
||||
hooks: true
|
||||
bin: true
|
||||
|
||||
marketplace:
|
||||
# apm's Claude marketplace mapper only emits description:/version: into the
|
||||
# compiled marketplace.json when set explicitly here (an override) — the
|
||||
# top-level apm.yml description:/version: above are NOT inherited into the
|
||||
# compiled output despite being used elsewhere (e.g. by `apm audit`).
|
||||
description: AI development skills for Claude Code and GitHub Copilot CLI — factory, design, implement, review, and cross-cutting workflows.
|
||||
version: 0.3.3
|
||||
version: 0.4.5
|
||||
owner:
|
||||
name: Defame1297
|
||||
email: defame1297@rkdr.net
|
||||
url: https://git.dev.rkdr.net/Defame1297/
|
||||
|
||||
# Default tag pattern used to resolve version ranges for each package.
|
||||
@@ -34,25 +79,25 @@ marketplace:
|
||||
- name: kyberforge
|
||||
description: Skills and agents for creating, maintaining, and managing a Claude Code / Copilot CLI plugin marketplace.
|
||||
source: ./plugins/kyberforge
|
||||
version: 1.4.1
|
||||
version: 1.6.0
|
||||
category: Developer Tools
|
||||
|
||||
- name: bin
|
||||
description: A place for things to be binned
|
||||
description: Skills for everyday AI-assisted development work that is not tied to a single tool, forge or language, and has not yet been split into a focused plugin.
|
||||
source: ./plugins/bin
|
||||
version: 1.1.1
|
||||
version: 1.1.5
|
||||
category: Utilities
|
||||
|
||||
- name: git
|
||||
description: Skills for working with Git — conventional commits, branch management, pull requests, and feature flow.
|
||||
description: Skills and agents for working with a local Git clone over the git wire protocol, and for authoring and running the pre-commit hooks that guard it.
|
||||
source: ./plugins/git
|
||||
version: 1.3.2
|
||||
version: 1.3.5
|
||||
category: Version Control
|
||||
|
||||
- name: gitea
|
||||
description: Skills for managing Gitea repositories — issues, pull requests, milestones, releases, and wikis.
|
||||
description: Skills and agents for working with a Gitea forge through its HTTP API — the forge's own objects, as distinct from the local git clone.
|
||||
source: ./plugins/gitea
|
||||
version: 1.3.3
|
||||
version: 1.3.6
|
||||
category: Version Control
|
||||
|
||||
- name: core
|
||||
@@ -64,11 +109,11 @@ marketplace:
|
||||
- name: mattpocock-skills
|
||||
description: Skills for Real Engineers — planning, TDD, architecture, and debugging workflows from Matt Pocock's .claude directory.
|
||||
source: mattpocock/skills
|
||||
version: "^1.2.0"
|
||||
version: "1.2.3"
|
||||
category: Productivity
|
||||
|
||||
- name: lint
|
||||
description: Skills and agents for configuring and running linters.
|
||||
source: ./plugins/lint
|
||||
version: 1.1.5
|
||||
version: 1.1.6
|
||||
category: Developer Tools
|
||||
|
||||
@@ -18,4 +18,4 @@ Three alternatives were rejected. Keeping the file-based fallback adds code comp
|
||||
|
||||
The file-based model also had a structural weakness: issues in `docs/issues/` were invisible from the Gitea UI, making it impossible to track work, assign milestones, or filter by label without opening the repo locally. Gitea provides all of that natively.
|
||||
|
||||
The "Provider-agnostic issue tracker" glossary entry in CONTEXT.md is updated in the same workstream to remove the file-based phase framing. The `providers/gitea/` adapter path described in ADR-0011 was never implemented — Gitea integration runs entirely via MCP, not a provider adapter.
|
||||
The "Provider-agnostic issue tracker" glossary entry in CONTEXT.md is updated in the same workstream to remove the file-based phase framing. (Amended 2026-08-17: the CONTEXT.md trim renamed that entry to **Issue**; it still records Gitea as this repo's canonical tracker and still tells skills to say "linked issue" generically.) The `providers/gitea/` adapter path described in ADR-0011 was never implemented — Gitea integration runs entirely via MCP, not a provider adapter.
|
||||
|
||||
@@ -9,6 +9,12 @@ deferred PR #85 review item to broaden that coverage, retroactively captures #84
|
||||
(since it was never recorded as a decision in its own right), and layers the expansion on top
|
||||
without reversing or weakening the original four rules.
|
||||
|
||||
**2026-08-17 amendment.** The CONTEXT.md section named above no longer holds that documentation.
|
||||
CONTEXT.md was cut back to a glossary and the prefilter's mechanics — the two-copy style layout,
|
||||
`vale-wrap.sh`, the `--config` argv defect, the rule inventory, and the 0-files-means-NOT-RUN
|
||||
fallback — moved to `docs/spec/gates.md`. Read that file, not CONTEXT.md, for the harness itself;
|
||||
this ADR still owns the scope decision.
|
||||
|
||||
**File scope stays the same.** `SKILL.md` plus agent files (`**/agents/*.md`,
|
||||
`**/*.agent.md`) only — matching the existing prefilter's globs. Skill-level
|
||||
`README.md` files and `plugin.json` manifests are not added: README.md files are navigational, not
|
||||
|
||||
@@ -56,6 +56,10 @@ new hand-maintained manifest format.
|
||||
that work through to merge.
|
||||
- `CONTEXT.md`'s "Plugin"/"Plugin marketplace" glossary entries were rewritten in issue #90 to
|
||||
describe the compiled-output model directly, rather than carrying a forward-pointer to this ADR.
|
||||
Superseded 2026-08-17: CONTEXT.md was cut back to one-line definitions, and the compiled-output
|
||||
model is now described in `docs/spec/architecture.md`. The same trim deleted the "lint plugin"
|
||||
entry cited under Considered options below; that pointer now reads `docs/spec/architecture.md`'s
|
||||
plugin scope table, which carries the repo-agnostic-versus-marketplace-specific argument.
|
||||
|
||||
## Considered options
|
||||
|
||||
@@ -66,7 +70,7 @@ maintenance in place unchanged.
|
||||
|
||||
**New standalone `plugins/apm/` plugin (rejected).** `plugins/lint/` was split out of `kyberforge`
|
||||
specifically because Vale tooling is generic and repo-agnostic, not holocron-marketplace-specific
|
||||
(see `CONTEXT.md`'s "lint plugin" entry) — the same argument applies to a generic `apm` CLI
|
||||
(see `docs/spec/architecture.md`'s plugin scope table) — the same argument applies to a generic `apm` CLI
|
||||
wrapper. The shipped `apm-install`/`apm-workflow` skills are, in fact, generic, repo-agnostic APM
|
||||
CLI documentation with no holocron-specific content, so a standalone `plugins/apm/` would have
|
||||
been a defensible split on artifact content alone. Rejected anyway, in favor of `kyberforge`,
|
||||
@@ -118,18 +122,21 @@ correction) sorted what they document into three buckets:
|
||||
match.
|
||||
- ADR-0016 (a narrower decision discovered while designing issue #89) turned out to gate how
|
||||
issue #90 had to re-author plugin-scope agents: `.apm/agents/*.agent.md` compiles verbatim to
|
||||
both Claude and Copilot, so those files carry only `name`/`description`/`model`/`source_keys` —
|
||||
existing dual-file `<name>.md`+`<name>.agent.md` pairs could not be raw-moved, only re-authored.
|
||||
both Claude and Copilot, so those files carry only the fields in the `apm-agent-allowlist` section
|
||||
of `plugins/kyberforge/.apm/skills/agent-audit/references/field-inventory.md` (as amended
|
||||
2026-08-14: `name`/`description`/`model`/`source_keys`/`disallowedTools`) — existing dual-file
|
||||
`<name>.md`+`<name>.agent.md` pairs could not be raw-moved, only re-authored.
|
||||
- Two follow-up issues tracked the remaining work: #89 (`skill-author`/`agent-author` routing
|
||||
adaptation — closed, merged in #93) and #90 (the actual repo conversion, which also deleted
|
||||
`plugin-author`/`marketplace-author` — tracked through to merge; treat #90's own state as the
|
||||
authority on whether it has landed, not this line).
|
||||
- **`displayName` is gone from all six compiled `plugin.json` files, and `owner.email` from the
|
||||
marketplace manifest — accepted, not overlooked.** `apm.yml` has no key that compiles to either,
|
||||
so the conversion dropped both: every `plugins/<name>/.claude-plugin/plugin.json` now carries
|
||||
- **`displayName` is gone from all six compiled `plugin.json` files — accepted, not overlooked.**
|
||||
`apm.yml` has no key that compiles to it: `synthesize_plugin_json_from_apm_yml`
|
||||
(`apm_cli/deps/plugin_parser.py`) emits only `name`, `version`, `description`, `author`,
|
||||
`license`, `homepage`, `repository` and `keywords`, and nothing in `plugin_manifest.py` adds
|
||||
`displayName` afterwards. So every `plugins/<name>/.claude-plugin/plugin.json` now carries
|
||||
`author`/`description`/`homepage`/`keywords`/`license`/`name`/`repository`/`version` (plus
|
||||
`mcpServers` for `bin`) and no `displayName`, and `.claude-plugin/marketplace.json`'s `owner`
|
||||
block is `{name, url}` only. Both fields are optional —
|
||||
`mcpServers` for `bin`) and no `displayName`. The field is optional —
|
||||
`plugins/kyberforge/docs/research/docs/claude-code-plugins/api-reference.md:14` lists
|
||||
`displayName` as `Required: No`, "Human-readable name shown in plugin manager" — which is why
|
||||
`claude plugin validate --strict` still passes on all six. The visible cost is that the plugin
|
||||
@@ -138,16 +145,31 @@ correction) sorted what they document into three buckets:
|
||||
`reinject_*` workaround of the kind ADR-0017's amendment reserves for fields apm strips on a
|
||||
factually wrong premise, and apm's premise here is simply that the key does not exist in its
|
||||
schema.
|
||||
- **`mattpocock-skills` is now version-pinned, and the pin is maintained by hand.** Pre-conversion
|
||||
the entry was `{"repo": "mattpocock/skills", "source": "github"}` — an unpinned reference that
|
||||
tracked the upstream default branch, so consumers got whatever was on it at install time. Root
|
||||
`apm.yml` now declares `version: "^1.2.0"` for it, which `apm pack` resolves and freezes into
|
||||
`.claude-plugin/marketplace.json` as `ref: v1.2.3` + an explicit `sha`. Consumers get a
|
||||
reproducible version instead of a moving target, which is the improvement; the cost is that
|
||||
nothing advances it. apm has no version-bump automation (established under "Versioning" in issue
|
||||
#90's plan), so picking up a new upstream release means a human editing the `version:` range in
|
||||
root `apm.yml` and re-running `apm pack`. Left un-bumped, the marketplace pins an ageing release
|
||||
indefinitely and silently.
|
||||
- **`owner.email` was dropped by mistake and has been restored (2026-08-14).** An earlier revision
|
||||
of this ADR listed `owner.email` alongside `displayName` as a field `apm.yml` "has no key that
|
||||
compiles to." That was wrong. `apm_cli/marketplace/yml_schema.py:186` defines
|
||||
`_AUTHOR_OBJECT_KEYS = frozenset({"name", "email", "url"})`, and an `email:` under root
|
||||
`apm.yml`'s `marketplace.owner` block was empirically confirmed to compile straight through into
|
||||
`.claude-plugin/marketplace.json`'s `owner`. The key is declared in root `apm.yml` again and the
|
||||
compiled `owner` block is `{name, email, url}`. Only `displayName` is a genuine schema gap; this
|
||||
one was a documentation error that removed working configuration.
|
||||
- **`mattpocock-skills` is pinned to an exact version, and the pin is advanced by hand.**
|
||||
Pre-conversion the entry was `{"repo": "mattpocock/skills", "source": "github"}` — an unpinned
|
||||
reference that tracked the upstream default branch, so consumers got whatever was on it at
|
||||
install time. The conversion first replaced that with `version: "^1.2.0"`, which was still not a
|
||||
pin: a caret range has nothing to freeze it, because there is no lockfile for
|
||||
`marketplace.packages[]`. `apm pack` re-resolved the range against upstream on **every** run, so
|
||||
an upstream `v1.2.4` would immediately invalidate the committed `ref`/`sha` and fail
|
||||
`apm-pack-check-clean` with exit 4 — blocking every push in the repo, triggered by a third party
|
||||
at an unrelated moment, with no local change to explain it. Root `apm.yml` therefore declares an
|
||||
exact `version: "1.2.3"`, which `apm pack` freezes into `.claude-plugin/marketplace.json` as
|
||||
`ref: v1.2.3` + an explicit `sha`. Two consequences, both intended: the committed ref/sha is
|
||||
genuinely reproducible and cannot move under the repo, and picking up a new upstream release is a
|
||||
deliberate act — a human edits the `version:` string in root `apm.yml` and re-runs `apm pack`.
|
||||
apm has no version-bump automation (established under "Versioning" in issue #90's plan), so an
|
||||
ageing pin is the accepted cost of a push gate that only fires on this repo's own changes.
|
||||
Note the pin does not make the entry offline-resolvable: an exact version still requires a
|
||||
`git ls-remote`, which is why two pre-push hooks need the network (see `AGENTS.md`).
|
||||
- **Caveat on "Status: executed" above:** issue #90's own execution comment flagged, before merge,
|
||||
that Claude Code's ability to actually load content out of `.apm/` was unverified — that caveat
|
||||
turned out to be a real defect, not a formality: the native installer has zero awareness of
|
||||
|
||||
@@ -22,13 +22,17 @@ Code and Copilot CLI targets. This is unlike:
|
||||
Because the agent primitive ships the same frontmatter unchanged to both harnesses, two
|
||||
concrete incompatibilities surface:
|
||||
|
||||
1. **`tools:`** — Claude Code expects a space-separated tool-name string; Copilot CLI expects a
|
||||
list drawn from its own alias vocabulary (`execute`/`read`/`edit`/`search`/`agent`/`web`). A
|
||||
value correct for one harness is wrong for the other.
|
||||
1. **`tools:`** — Claude Code expects tool names drawn from its own vocabulary, as a
|
||||
comma-separated string or a YAML list (`agent-definition.md:37`); Copilot CLI expects a list
|
||||
drawn from a different alias vocabulary (`execute`/`read`/`edit`/`search`/`agent`/`web`). The
|
||||
incompatibility is the vocabulary, not the punctuation: a value correct for one harness names
|
||||
tools the other does not have.
|
||||
2. **Claude-only knobs with no Copilot equivalent** — `isolation`, `maxTurns`, `effort`,
|
||||
`memory`, `permissionMode`. Writing any of these means Copilot's copy carries frontmatter
|
||||
keys it doesn't recognize at all. Whether Copilot's agent loader ignores unknown keys or
|
||||
errors on them is unconfirmed by research.
|
||||
errors on them is unconfirmed by research. *(Still unconfirmed as of the 2026-08-14 amendment
|
||||
below, which admits `disallowedTools` as an explicitly accepted risk rather than by resolving
|
||||
this question.)*
|
||||
|
||||
## Decision
|
||||
|
||||
@@ -36,6 +40,9 @@ At **plugin scope only** (destination package has an `apm.yml` at its root — a
|
||||
package compiled via `apm compile`), `.apm/agents/<name>.agent.md` carries only `name`,
|
||||
`description`, `model`, and the prose body. No `tools:` field, no Claude-only fields, at all.
|
||||
|
||||
*(Narrowed by the 2026-08-14 amendment below: `disallowedTools` is admitted as a fifth allowed
|
||||
field. `tools:` and every other Claude-only knob remain excluded on the reasoning given here.)*
|
||||
|
||||
Absent `tools:` means inherit-all-tools on both harnesses — the one value that is never wrong
|
||||
on either target, unlike a present, harness-specific value that is guaranteed wrong on at least
|
||||
one of them.
|
||||
@@ -70,11 +77,84 @@ harness. Tracking the breakage doesn't prevent it, and the chosen decision alrea
|
||||
equivalent visibility (a SUGGESTION finding) without ever shipping the wrong value in the first
|
||||
place.
|
||||
|
||||
## Amendment (2026-08-14): the write fence comes back as a denylist
|
||||
|
||||
The decision above generalised from `tools:` to "no tool restriction at all". That over-reached.
|
||||
The unportability argument is specific to the **allowlist**: Claude Code reads `tools:` as a
|
||||
delimited string of its own tool names, Copilot CLI reads it as a list drawn from its
|
||||
alias vocabulary (`execute`/`read`/`edit`/`search`/`agent`/`web`), so one value is wrong on one
|
||||
harness. That reasoning stands, and `tools:` stays out of every plugin-scope agent.
|
||||
|
||||
A **denylist** has no such conflict. The evidence for that splits three ways, and this amendment
|
||||
states which part is which rather than asserting the whole as settled.
|
||||
|
||||
**Confirmed — Claude Code honours it for plugin subagents.**
|
||||
`plugins/kyberforge/docs/research/docs/claude-code-plugins/agent-definition.md:39` documents
|
||||
`disallowedTools` as a "Denylist applied before `tools`… Takes precedence over `tools`", and — the
|
||||
part that matters here — it is **not** in that document's plugin-subagent ignore list. Line 99
|
||||
names exactly three fields plugin agents silently ignore: `hooks`, `mcpServers`, `permissionMode`.
|
||||
`disallowedTools` is absent from that list. Claude Code is also the harness where the fence is
|
||||
actually wanted, so the field earns its place on this evidence alone.
|
||||
|
||||
**Inferred — the field is very likely inert on Copilot CLI, but by analogy, not by documentation.**
|
||||
`plugins/kyberforge/docs/research/docs/github-copilot-plugins/troubleshooting.md:50` and `:53`
|
||||
record Copilot *silently ignoring* two agent frontmatter fields it does not process (`mcp-servers`
|
||||
and `metadata` outside the cloud runtime) rather than erroring on them. That is a documented
|
||||
tolerance for *known-but-unprocessed* keys, which is adjacent to, not identical to, tolerance for
|
||||
an *unknown* key. No stronger evidence exists: a sweep of the vendored Copilot corpus
|
||||
(`agent-definition.md`, `api-reference.md`, `troubleshooting.md`, `configuration.md`) documents
|
||||
unknown-key handling nowhere.
|
||||
|
||||
**Unverified — Copilot's loader behaviour on an unrecognised key.** Context item 2 above says this
|
||||
is unconfirmed by research and that remains true; nothing found since changes it. An earlier
|
||||
revision of this amendment claimed "an unrecognised frontmatter key is inert" as settled fact and
|
||||
attributed it to apm's verbatim-copy behaviour. That attribution was a non-sequitur — verbatim copy
|
||||
describes what *apm* does at compile time and says nothing about what *Copilot* does at load time —
|
||||
and the claim contradicted this ADR's own Context section.
|
||||
|
||||
**So this is an accepted risk, stated as one.** Blast radius if the inference is wrong and Copilot
|
||||
errors on the key: the three affected plugin-scope agents fail to load under Copilot CLI. It is
|
||||
loud, not silent; it is confined to three agents in three plugins; no other primitive and no Claude
|
||||
Code path is affected; and the remedy is a one-line frontmatter deletion. What the denylist shape
|
||||
*does* rule out categorically — independent of loader behaviour — is the failure mode that motivated
|
||||
dropping `tools:` in the first place: a denied name the other harness does not recognise denies
|
||||
nothing, so a mis-shaped value can never grant or misroute a capability. The risk is a load failure,
|
||||
never a silent over-grant. That asymmetry is why the same verbatim copy that makes `tools:`
|
||||
unshippable makes `disallowedTools` worth shipping.
|
||||
|
||||
So the read-only orchestrator agents regain their write fence: `gitea-orchestrate`,
|
||||
`apm-orchestrate` and `lint-runner` each carry `disallowedTools: Edit, Write, NotebookEdit` plus
|
||||
explicit prose in the body stating the agent does not edit files. `git-orchestrate` is deliberately
|
||||
excluded — it legitimately declared `edit` before the conversion and still needs to write.
|
||||
|
||||
**Residual — the fence is partial, and the prose is doing more of the work than the field is.**
|
||||
`disallowedTools: Edit, Write, NotebookEdit` denies exactly those three tools. It does not deny
|
||||
`Bash`, and at plugin scope these agents carry no `tools:` and therefore inherit it, so
|
||||
`bash -c 'echo … > f'` remains unfenced by frontmatter. Only the body prose covers that path. This
|
||||
is not a regression introduced here — the pre-conversion `tools:` allowlists also granted `Bash`,
|
||||
so the shell route was open then too — but the ADR should not credit the mechanism with more than
|
||||
it delivers. Closing it would need a `disallowedTools` entry for `Bash`, which these agents cannot
|
||||
take because they legitimately shell out.
|
||||
|
||||
Net position: the allowlist stays dropped for the reason originally given, and the denylist is
|
||||
admitted as the portable-by-construction half of what was lost. It restores a real, Claude-Code-
|
||||
confirmed write fence against the tool-call path, not a complete write sandbox. The consequence
|
||||
below is narrowed accordingly.
|
||||
|
||||
Enforcement follows the decision: `agent-audit`'s plugin-scope validator reads its allowlist as
|
||||
data from the `apm-agent-allowlist` section of
|
||||
`plugins/kyberforge/.apm/skills/agent-audit/references/field-inventory.md`, and that line now reads
|
||||
`name description model source_keys disallowedTools`. `disallowedTools` also stays in that file's
|
||||
`claude-code-only-fields` list, which is not a contradiction — that list governs whether a field
|
||||
may cross the CC/Copilot boundary in a real project/user-scope *pair*, a different question from
|
||||
whether a field is safe under verbatim copy in a single vendor-neutral file.
|
||||
|
||||
## Consequences
|
||||
|
||||
- Every plugin-scope APM agent loses per-agent tool restriction and any Claude-only capability
|
||||
- Every plugin-scope APM agent loses per-agent tool *allowlisting* and any Claude-only capability
|
||||
(isolation, maxTurns, effort, memory, permissionMode) until APM ships a real per-target
|
||||
integrator for the agent primitive. This is a known, accepted regression, not an oversight.
|
||||
Tool **denial** is not part of that loss — see the 2026-08-14 amendment above.
|
||||
- **ADR-0005 is partially superseded** — its plugin-scope clause ("directory containing
|
||||
`plugin.json` is plugin scope → both files land in `<root>/agents/`") no longer applies.
|
||||
Plugin scope is now "directory containing `apm.yml` → single vendor-neutral file lands in
|
||||
@@ -86,7 +166,9 @@ place.
|
||||
lists from `references/field-inventory.md` rather than hardcoding them, with a `source_keys`
|
||||
provenance chain — survives and is reused. Only the *content shape* changes for plugin scope:
|
||||
`field-inventory.md` shifts from two side-by-side CC-only/Copilot-only blocklists to one
|
||||
vendor-neutral allowlist (`name`/`description`/`model`/`source_keys` — the last for provenance
|
||||
tracking, validated separately by `validate-provenance.sh` against `sources.md`, not a
|
||||
provider-specific field) for plugin-scope agents, while
|
||||
continuing to serve its original two-blocklist role for project/user-scope validation.
|
||||
vendor-neutral allowlist for plugin-scope agents, while continuing to serve its original
|
||||
two-blocklist role for project/user-scope validation. That file's `apm-agent-allowlist` section
|
||||
is the authoritative list and is read as data by `validate.sh`; as amended on 2026-08-14 it holds
|
||||
`name`/`description`/`model`/`source_keys`/`disallowedTools` — `source_keys` for provenance
|
||||
tracking, validated separately by `validate-provenance.sh` against `sources.md` rather than being
|
||||
a provider-specific field, and `disallowedTools` per the amendment above.
|
||||
|
||||
@@ -86,7 +86,11 @@ governance status as `.claude-plugin/plugin.json`/`marketplace.json`:
|
||||
`.pre-commit-config.yaml` as hook id `check-plugin-content-sync` by a parallel workstream on
|
||||
issue #90) — the same enforcement model `check-manifests.sh` already applies to the other
|
||||
compiled-output category. `--check` alone is not the gate: the script requires either `--all` or
|
||||
an explicit list of plugin directories, and run bare it prints usage and exits 1.
|
||||
an explicit list of plugin directories, and run bare it prints usage and exits 1. `--all` derives
|
||||
its work list from `marketplace.json`, a generated file, so it asserts its own coverage against
|
||||
that list: it fails if it verified fewer plugins than the marketplace declares, not merely if it
|
||||
verified none. A listed plugin whose `.apm/` has gone missing is skipped by the per-plugin sync
|
||||
and would otherwise let the gate report success over a shrinking work list.
|
||||
- Verified two ways before landing: `claude plugin validate --strict` passes on all 6 real
|
||||
(non-scratch) plugin directories, and a live behavioral test
|
||||
(`claude --plugin-dir plugins/kyberforge -p "list your skills and agents"`) against the real
|
||||
@@ -98,18 +102,29 @@ governance status as `.claude-plugin/plugin.json`/`marketplace.json`:
|
||||
|
||||
## Considered options
|
||||
|
||||
**Patch `plugin.json`'s `skills`/`agents`/`commands`/`hooks` fields to point directly at `.apm/`
|
||||
paths (rejected).** Claude Code's manifest schema documents these as legitimate override fields
|
||||
that accept custom paths —
|
||||
`plugins/kyberforge/docs/research/docs/claude-code-plugins/configuration.md` shows a real example
|
||||
(`"skills": "./custom/skills/"`, `"agents": ["./custom/agents/reviewer.md"]`), so the host side of
|
||||
this would work. Rejected because apm's compiler is not a passive pass-through: `build_plugin_manifest`
|
||||
unconditionally strips these keys from every manifest it generates, on the stated assumption that
|
||||
convention directories are always host-auto-discovered and therefore never need an explicit
|
||||
pointer. Honoring this option would mean post-processing apm's compiled output on every
|
||||
`apm pack` run to re-inject fields apm actively removes — fighting a stable, intentional apm code
|
||||
path indefinitely — rather than reusing `plugin_exporter.py`'s bundle-export mapping, which already
|
||||
does the right thing and only needed its output redirected to a path the installer reads.
|
||||
**Patch `plugin.json`'s content-pointer fields to point directly at `.apm/` paths (rejected).**
|
||||
Claude Code's manifest schema documents these as legitimate override fields that accept custom
|
||||
paths — `plugins/kyberforge/docs/research/docs/claude-code-plugins/configuration.md` shows a real
|
||||
example (`"skills": "./custom/skills/"`, `"agents": ["./custom/agents/reviewer.md"]`), so the host
|
||||
side of this would work. Rejected because apm never emits such a pointer and would have to be
|
||||
worked around on every run to make it do so.
|
||||
|
||||
Be precise about the mechanism, because an earlier revision of this ADR overstated it. apm 0.28.0's
|
||||
`build_plugin_manifest` (`apm_cli/core/plugin_manifest.py`) does carry a strip loop, but its field
|
||||
list is `("agents", "skills", "commands", "instructions")` — `hooks` is **not** in it, and
|
||||
`instructions` **is**, which this ADR previously did not mention. More to the point, that loop can
|
||||
never fire: the manifest it operates on comes from `synthesize_plugin_json_from_apm_yml`
|
||||
(`apm_cli/deps/plugin_parser.py`), which only ever emits `name`, `version`, `description`,
|
||||
`author`, `license`, `homepage`, `repository` and `keywords`. The pointer fields are absent from
|
||||
apm's output because `apm.yml` has no schema for them, not because apm actively removes them — the
|
||||
`pop` loop is defensive dead code against a manifest shape apm does not produce.
|
||||
|
||||
The rejection is unaffected by that correction, only its framing. Honoring this option would still
|
||||
mean post-processing apm's compiled output on every `apm pack` run to add fields apm's schema has
|
||||
no way to express, rather than reusing `plugin_exporter.py`'s bundle-export mapping, which already
|
||||
does the right thing and only needed its output redirected to a path the installer reads. What it
|
||||
is *not* is a fight against a load-bearing apm code path — the honest statement is that apm has no
|
||||
input for these fields, and inventing one downstream is a workaround this ADR did not need.
|
||||
|
||||
**Point `marketplace.json`'s `source:` at `apm pack`'s `build/<name>-<version>/` output directly
|
||||
(rejected).** Would reuse the bundle exporter's correct mapping without adding a new script.
|
||||
@@ -121,36 +136,50 @@ source and scans directories; it does not execute a package manager's build comm
|
||||
Copying the relevant subset back to the stable `plugins/<name>/` path — where `marketplace.json`
|
||||
already points — needed no change to the marketplace source model at all.
|
||||
|
||||
## Amendment (2026-08-13): `mcpServers` is narrowly reinjected into Copilot's `plugin.json`
|
||||
## Amendment (2026-08-13, revised 2026-08-14): Copilot's `plugin.json` gets an `mcpServers` *path*
|
||||
|
||||
PR #95's review (a follow-on to this same issue #90 workstream) found a second field apm's
|
||||
compiler strips for the Copilot ecosystem: `build_plugin_manifest` unconditionally removes
|
||||
`mcpServers` from every Copilot-ecosystem `plugin.json`, its docstring stating the field is "not
|
||||
part of the Copilot plugin manifest schema." That claim is contradicted by this repo's own
|
||||
researched documentation — `plugins/kyberforge/docs/research/docs/github-copilot-plugins/
|
||||
configuration.md:49` documents `mcpServers` as a valid, optional `plugin.json` field for Copilot.
|
||||
compiler drops for the Copilot ecosystem: `build_plugin_manifest` runs
|
||||
`manifest.pop("mcpServers", None)` on every Copilot-ecosystem `plugin.json`, its docstring stating
|
||||
the field is "not part of the Copilot plugin manifest schema." That claim is contradicted by this
|
||||
repo's own researched documentation —
|
||||
`plugins/kyberforge/docs/research/docs/github-copilot-plugins/configuration.md:49` documents
|
||||
`mcpServers` as a valid, optional `plugin.json` field, typed **"string or object — MCP server
|
||||
config path or inline definitions."**
|
||||
|
||||
This is not the same situation "Considered options" above rejected. That rejection concerned
|
||||
fields apm strips *correctly*, on a stable and accurate premise: convention directories
|
||||
(`skills/`, `agents/`, `commands/`) are host-auto-discovered, so an explicit pointer is redundant
|
||||
by design. Here, apm's own stated justification for stripping `mcpServers` is factually wrong
|
||||
against documented Copilot behavior — there is no host-auto-discovery mechanism that makes an
|
||||
explicit `mcpServers` declaration redundant, the way there is for skills/agents/commands. Applying
|
||||
the same "don't fight a stable, intentional apm code path" reasoning here would mean shipping a
|
||||
plugin manifest known to be missing a field Copilot actually reads.
|
||||
This is not the same situation "Considered options" above rejected. There, apm emits no pointer
|
||||
because its schema has no input for one and the host auto-discovers the directories anyway, so
|
||||
nothing is missing. Here a field Copilot actually reads is actively removed on a premise that is
|
||||
wrong against documented Copilot behavior, and there is no auto-discovery mechanism that makes it
|
||||
redundant. Shipping the manifest as apm produces it would ship a manifest known to be incomplete.
|
||||
|
||||
Given that, `scripts/sync-plugin-content.sh`'s `reinject_mcp_servers()`, called from `sync_one()`,
|
||||
narrowly re-injects `mcpServers` into `.github/plugin/plugin.json` after `apm pack` runs, sourced
|
||||
from the plugin's own `.mcp.json`, and only when it declares at least one server — matching apm's
|
||||
own Claude-ecosystem builder, which omits the field entirely rather than emitting
|
||||
`mcpServers: {}`. **Both modes re-inject**, not just real syncs: real mode writes into the plugin
|
||||
root directly, `--check` into its throwaway copy first, so the manifest diff compares against the
|
||||
same content a real sync would actually produce (see the script's own header). A check-mode
|
||||
re-injection is what keeps `--check` from reporting permanent phantom drift on every plugin that
|
||||
ships an `.mcp.json`. This is scoped to one field found to be incorrectly stripped, not a
|
||||
reversal of the broader position above: the rejection of patching
|
||||
`skills`/`agents`/`commands`/`hooks` pointers still holds, since apm's premise for stripping those
|
||||
remains accurate.
|
||||
`scripts/sync-plugin-content.sh`'s `reinject_mcp_servers()`, called from `sync_one()`, therefore
|
||||
sets `mcpServers` on `.github/plugin/plugin.json` after `apm pack` runs — to the **string
|
||||
`".mcp.json"`**, the path form of the documented type, not the resolved server objects. Only when
|
||||
the plugin's `.mcp.json` declares at least one server, matching apm's own Claude-ecosystem builder,
|
||||
which omits the field entirely rather than emitting `mcpServers: {}`.
|
||||
|
||||
**The payload is a path because an inlined object is a credential-leak path.** The original
|
||||
implementation copied `.mcp.json`'s resolved `mcpServers` object into the manifest with `jq`. That
|
||||
route bypasses apm's own `_sanitize_mcp_servers()` (`apm_cli/core/plugin_manifest.py`), which
|
||||
strips credential keys and redacts secret values out of `.mcp.json` precisely because — in its own
|
||||
words — "copying them verbatim into a committed `plugin.json` would exfiltrate them into the
|
||||
distributed artefact." Today's `.mcp.json` files here carry no `env` block, so nothing leaked; the
|
||||
first one that did would have written a live token into a tracked, published manifest, with the
|
||||
sanitizer sitting one code path away and never invoked. A path reference cannot carry a secret at
|
||||
all: the manifest names a file, and resolution happens in the host at load time. This also matches
|
||||
apm's documented posture for MCP secrets — `microsoft-apm/configuration.md:96-98` requires `${VAR}`
|
||||
indirection so secrets are "never committed to the manifest."
|
||||
|
||||
**Both modes re-inject**, not just real syncs: real mode writes into the plugin root directly,
|
||||
`--check` into its throwaway copy first, so the manifest diff compares against the same content a
|
||||
real sync would actually produce (see the script's own header). A check-mode re-injection is what
|
||||
keeps `--check` from reporting permanent phantom drift on every plugin that ships an `.mcp.json`.
|
||||
|
||||
This remains scoped to one field found to be incorrectly dropped. It does not reopen the
|
||||
content-pointer option rejected above: those fields stay absent because apm has no schema input for
|
||||
them and the host needs no pointer, which is a different situation from a documented field being
|
||||
actively removed.
|
||||
|
||||
Consequence: if a future apm release corrects the Copilot `mcpServers` omission, `reinject_mcp_servers()`
|
||||
and its call site become dead code and should be deleted — nothing else in this ADR depends on the
|
||||
@@ -166,13 +195,15 @@ A function name is stable enough to grep for; a line number in an ADR is stale b
|
||||
As originally executed, `sync-plugin-content.sh` wrote the merged hooks file to
|
||||
`plugins/<name>/hooks.json`. That path is scanned by nothing. Claude Code convention-scans
|
||||
`hooks/hooks.json`, and the "Plugin Directory Layout" table this ADR's own root-cause analysis
|
||||
quotes above says so on the same line it says "All content directories must be at the plugin root,
|
||||
not inside `.claude-plugin/`"
|
||||
(`plugins/kyberforge/docs/research/docs/claude-code-plugins/configuration.md:100`). The
|
||||
implementation read "at the plugin root" and dropped the file there; the row it was reading names
|
||||
`hooks/hooks.json`. So this ADR shipped with the contract quoted correctly in its diagnosis and
|
||||
violated in its output — the flat mirror bridged skills and agents into discovery and left hooks
|
||||
exactly as undiscoverable as before the fix.
|
||||
quotes above says so:
|
||||
`plugins/kyberforge/docs/research/docs/claude-code-plugins/configuration.md:100` is the row naming
|
||||
`hooks/hooks.json`, six lines below the table's preamble at `:94` — "All content directories must
|
||||
be at the plugin root, not inside `.claude-plugin/`". The two are not the same line; an earlier
|
||||
revision of this amendment said they were. The implementation read the preamble's "at the plugin
|
||||
root" and dropped the file there, without reading the row that names the path. So this ADR shipped
|
||||
with the contract quoted correctly in its diagnosis and violated in its output — the flat mirror
|
||||
bridged skills and agents into discovery and left hooks exactly as undiscoverable as before the
|
||||
fix.
|
||||
|
||||
The merged file therefore moves to `plugins/<name>/hooks/hooks.json`. A root-level `hooks.json`
|
||||
left over from a prior sync is stale output: a real sync deletes it, `--check` reports it as
|
||||
@@ -184,9 +215,99 @@ only those two grow a mirrored hooks file at all.
|
||||
This does **not** reopen the "patch `plugin.json` pointer fields" option rejected above. The move
|
||||
needs no `hooks` pointer in `plugin.json`: `hooks/hooks.json` *is* the convention path, so the
|
||||
host finds it by auto-discovery, exactly as it finds `skills/` and `agents/`. The rejection stands
|
||||
for the reason it was made — apm's `build_plugin_manifest` strips pointer fields unconditionally
|
||||
and is right to, because convention directories need no pointer. Writing to the convention path is
|
||||
what makes that premise true here rather than something to fight.
|
||||
for the reason it was made, once stated accurately — apm emits no pointer field for any of these,
|
||||
because `apm.yml` has no key that produces one, and none is needed when content sits at the
|
||||
convention path. (`hooks` was never in `build_plugin_manifest`'s strip list at all; see the
|
||||
corrected mechanism note under "Considered options".) Writing to the convention path is what makes
|
||||
the no-pointer premise true here rather than something to work around.
|
||||
|
||||
Read "the host finds it by auto-discovery" above as **Claude Code**, not both hosts. Copilot has no
|
||||
default for `hooks` and so discovers none — a real gap, examined and deliberately left open in the
|
||||
next amendment.
|
||||
|
||||
## Amendment (2026-08-14): no `hooks` pointer is re-injected for Copilot — the gap stays documented
|
||||
|
||||
PR #95's review found a third field, and it looks like the `mcpServers` amendment's exact twin:
|
||||
`plugins/kyberforge/docs/research/docs/github-copilot-plugins/configuration.md:47` types `hooks` as
|
||||
a `plugin.json` field, **"string or object"**, with **no default** — so Copilot has no convention
|
||||
path to scan — and `jq 'has("hooks")'` returns `false` for all six `plugins/*/.github/plugin/plugin.json`.
|
||||
Copilot therefore resolves **zero hooks from every plugin in this repo**. The facts are not in
|
||||
dispute; the remedy is.
|
||||
|
||||
State the mechanism correctly first, because it differs from `mcpServers` and the amendment above
|
||||
depends on that distinction. `mcpServers` is *actively removed* — `build_plugin_manifest` runs
|
||||
`manifest.pop("mcpServers", None)` on every Copilot manifest. `hooks` was **never in that strip
|
||||
list** (its field list is `("agents", "skills", "commands", "instructions")`, and the loop is dead
|
||||
code besides — see "Considered options"). This is an absence apm never fills, not a removal to
|
||||
reverse.
|
||||
|
||||
**Decision: do not re-inject. Document the gap.** The `mcpServers` exception was granted on three
|
||||
conditions, and `hooks` meets only two of them:
|
||||
|
||||
1. *A documented host schema field.* Met — `hooks` is in Copilot's own field table.
|
||||
2. *apm has no input that produces it.* Met — `apm.yml` has no key for it.
|
||||
3. *The payload is correct for the host regardless of content.* **Not met**, and this is the whole
|
||||
difference. `.mcp.json` is one host-agnostic format that both ecosystems read, so the string
|
||||
`".mcp.json"` is a true statement about the file no matter what is in it. Hooks have no such
|
||||
shared format: Claude Code reads
|
||||
`{"hooks": {"PreToolUse": [{"matcher": ..., "hooks": [...]}]}}` while Copilot requires
|
||||
`{"version": 1, "hooks": {"sessionStart": [{"type": "command", "bash": ..., "powershell": ...}]}}`
|
||||
— a mandatory `version`, lowercase and differently-named lifecycle events, and per-shell script
|
||||
keys. apm's exporter merges `.apm/hooks/*.json` into **exactly one** `hooks.json` with no
|
||||
per-target shaping (`_collect_hooks_from_apm`, `apm_cli/bundle/plugin_exporter.py`), and that one
|
||||
file also sits at Claude Code's convention path, where Claude Code will read it whatever it
|
||||
contains. So there is exactly one file and two incompatible readers of it.
|
||||
|
||||
A `hooks` pointer would therefore assert that a Claude-shaped file is Copilot-shaped. That trades an
|
||||
*incomplete* manifest for a *wrong* one, which is the opposite of the `mcpServers` amendment's
|
||||
reasoning ("shipping the manifest as apm produces it would ship a manifest known to be incomplete").
|
||||
|
||||
The "it changes nothing today, so it is zero-risk and correct-by-construction for the first real
|
||||
hook" argument does not survive the same check, in both halves. It is not inert today: both
|
||||
`hooks/hooks.json` files are `{"hooks": {}}`, which lacks the `version: 1` Copilot's schema
|
||||
requires, so a pointer would name a file invalid against the schema it is being pointed at from —
|
||||
a change from "declares no hooks" to "declares hooks, at an invalid file". And it is not
|
||||
correct-by-construction later: whoever writes the first real hook writes it in one of the two
|
||||
shapes, and the pointer is wrong in the Claude-shaped case (the case that actually happens, since
|
||||
Claude Code auto-discovers the same file and is what these hooks are authored against) while the
|
||||
Copilot-shaped case breaks Claude Code instead. No content makes both readers correct.
|
||||
|
||||
What would change this decision is upstream, not local: apm emitting a per-target hooks file (at
|
||||
which point a pointer names a file genuinely shaped for its reader), or the two hook schemas
|
||||
converging. Until then the honest artifact is a documented gap, recorded for authors in
|
||||
`plugins/kyberforge/docs/hooks.md` and pinned by a test asserting the Copilot manifest carries no
|
||||
`hooks` key — so that adding one is a deliberate act that has to confront the schema mismatch,
|
||||
rather than a plausible-looking one-liner nobody re-derives.
|
||||
|
||||
This does not weaken the `mcpServers` amendment. That exception was narrow on purpose, and this is
|
||||
what its third condition was for.
|
||||
|
||||
## Amendment (2026-08-14): symlinks under `.apm/` are dropped, and are now reported
|
||||
|
||||
apm's bundle exporter filters symlinks out of the bundle entirely — `f.is_file() and not
|
||||
f.is_symlink()` in `_collect_flat` and `_collect_recursive`, and the same test in
|
||||
`_collect_hooks_from_apm` (`apm_cli/bundle/plugin_exporter.py`). It emits no warning. A symlink
|
||||
placed under a plugin's `.apm/` therefore never reaches the mirror, and until now nothing said so.
|
||||
|
||||
This was **silent content loss, not drift**, and that distinction is why no existing gate caught it.
|
||||
Every other check in `sync-plugin-content.sh` compares the live mirror against a freshly synced
|
||||
copy — and both sides are built from that same bundle. The symlink is absent from both, they agree,
|
||||
and `--check` exits 0. There is no mismatch to detect, only an absence with nothing left to
|
||||
mismatch against. Reproduced on a fixture: `ln -s real.md link.md` under `.apm/skills/hello/`
|
||||
produced a mirror with no `link.md` and a `--check` at exit 0.
|
||||
|
||||
`check_apm_symlinks()` therefore reads the `.apm/` **source** tree directly — the only place the
|
||||
loss is visible — and reports each symlink in both modes, failing the run. It is reported rather
|
||||
than resolved: dereferencing and copying the target would make a real sync emit content the bundle
|
||||
does not contain, which is precisely the "reimplement apm's mapping outside apm" this ADR rejects.
|
||||
Telling the author is the in-contract half.
|
||||
|
||||
The scan covers only the `.apm/` directories apm's exporter actually reads
|
||||
(`agents`, `skills`, `prompts`, `commands`, `instructions`, `extensions`, `hooks`), and carves out
|
||||
`<category>/<name>/tests` to match the mirror's own exclusion — that subtree is not mirrored whether
|
||||
or not it holds a symlink, so nothing is lost there. The carve-out is depth-scoped for the same
|
||||
reason the `tests/` exclusion is: a symlink under `assets/templates/tests` sits in content the
|
||||
mirror does carry, and is reported.
|
||||
|
||||
## Consequences
|
||||
|
||||
@@ -208,7 +329,9 @@ what makes that premise true here rather than something to fight.
|
||||
stands; this ADR fixes the second, previously-unverified half.
|
||||
- `CONTEXT.md`'s "Plugin" and "Plugin marketplace" glossary entries are updated to describe the
|
||||
flat mirror as a second compiled-output category, alongside the existing
|
||||
`.claude-plugin/plugin.json`/`marketplace.json` description.
|
||||
`.claude-plugin/plugin.json`/`marketplace.json` description. Superseded 2026-08-17: CONTEXT.md was
|
||||
cut back to one-line definitions and no longer describes either compiled-output category;
|
||||
`docs/spec/architecture.md` is where the mirror is documented.
|
||||
- A future apm release that ships a native `.apm/`-aware plugin.json compiler (closing this gap
|
||||
upstream) would let `sync-plugin-content.sh` and its drift gate be deleted outright — nothing in
|
||||
this ADR's decision depends on the flat mirror existing beyond satisfying the current installer's
|
||||
|
||||
149
docs/adr/0018-repo-consumes-its-own-plugins-through-apm.md
Normal file
149
docs/adr/0018-repo-consumes-its-own-plugins-through-apm.md
Normal file
@@ -0,0 +1,149 @@
|
||||
# This repo installs its own plugins through apm, not Claude Code's native plugin install
|
||||
|
||||
ADR-0015 moved plugin **authoring** to apm; ADR-0017 added the flat content mirror that keeps the
|
||||
authored `.apm/` tree discoverable by hosts that install natively. Both are about producing the
|
||||
marketplace. This ADR is about consuming it: how the plugins get onto the machine this repo is
|
||||
worked on.
|
||||
|
||||
**Status: executed (2026-08-14).** All six packages are installed into `/root/ai-development` by
|
||||
`apm install`; the six native project-scope installs (`claude plugin uninstall <name>@holocron
|
||||
--scope project`) are gone and `.claude/settings.json`'s `enabledPlugins` block is empty.
|
||||
|
||||
## Context
|
||||
|
||||
Until now the repo consumed its own output the same way any user would: `claude plugin install
|
||||
<name>@holocron`, six plugins enabled per-project in `.claude/settings.json`, the `holocron`
|
||||
marketplace registered in `~/.claude/plugins/known_marketplaces.json` with `autoUpdate: true`.
|
||||
That worked. It also meant the repo's dogfooding stopped one layer short of the tooling it
|
||||
publishes: `kyberforge` ships `apm-workflow` and `apm-install` skills describing an install path
|
||||
the repo itself did not take.
|
||||
|
||||
apm supports both scopes. `apm install --global` deploys to `~/.claude/`; plain `apm install`
|
||||
deploys to the project. Global was rejected deliberately — the switch should be provable in one
|
||||
repo before it changes how every other project on the machine resolves its skills.
|
||||
|
||||
## Decision
|
||||
|
||||
Root `apm.yml` declares all six packages under `dependencies.apm`, each as a `git:`/`path:` object
|
||||
against the holocron remote:
|
||||
|
||||
```yaml
|
||||
dependencies:
|
||||
apm:
|
||||
- git: git@git.dev.rkdr.net:Defame1297/holocron.git
|
||||
path: plugins/core
|
||||
```
|
||||
|
||||
`apm install` deploys them to `.claude/skills/<name>/` and `.claude/agents/<name>.md`.
|
||||
|
||||
Three sub-decisions inside that:
|
||||
|
||||
- **Object form over the `<name>@holocron` marketplace alias.** The alias is shorter and apm
|
||||
resolves it correctly (verified end-to-end against this remote), but it first requires
|
||||
`apm marketplace add`, which writes to `~/.apm/marketplaces.json` — user scope, outside the
|
||||
repo, and absent on a fresh clone. The object form needs nothing beyond the committed manifest.
|
||||
- **Remote source over local path.** apm accepts `path: /root/ai-development/plugins/<name>` as a
|
||||
local dependency, which would make the working tree live instantly. Rejected: it erases the
|
||||
distinction between editing a skill and shipping one, which is the entire point of having a
|
||||
marketplace. The remote form keeps the repo running the same released content every other
|
||||
consumer gets.
|
||||
- **Unpinned against the default branch.** Parity with the `autoUpdate: true` the native install
|
||||
had. apm warns on every install (`6 dependencies unpinned`); accepted knowingly. Pinning is a
|
||||
per-entry `ref:` away once the repo tags releases per package — today `git tag` lists one tag
|
||||
total, so there is nothing meaningful to pin to.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Skills gain an unnamespaced name.** apm deploys plain project skills, so `git:git-commits` also
|
||||
answers to `git-commits` and `kyberforge:skill-audit` to `skill-audit`. This is not configurable —
|
||||
a project skill has no plugin to prefix. `AGENTS.md` and `CONTEXT.md` are updated to name the bare
|
||||
form, which is what apm deploys and the only form a repo consuming holocron through apm gets.
|
||||
|
||||
**Correction (2026-08-14): the namespaced form did not stop resolving.** *Superseded by the
|
||||
2026-08-17 correction below: the machine state this cites is no longer present. Both are kept
|
||||
because the pair is the finding — read neither as current.* An earlier revision of
|
||||
this consequence said every `<plugin>:<skill>` reference "was stale the moment the switch landed",
|
||||
and `AGENTS.md`/`CONTEXT.md` were written to match. That contradicts the "User scope is untouched,
|
||||
deliberately" consequence below, and the contradiction resolves against it: `~/.claude.json` still
|
||||
enables `core`, `git`, `gitea`, `kyberforge` and `lint` at user scope, so both names are live at
|
||||
once and a working `gitea:gitea-prs` is the user-scope copy answering. That doubling is the same
|
||||
"present twice under two names" outcome the "Keeping both install paths" alternative was rejected
|
||||
for — reached by leaving user scope alone rather than by adopting it, which is why it is a
|
||||
consequence to record rather than a decision to revisit. Prefer the bare name regardless: it
|
||||
survives those user-scope installs eventually being converted, and the namespaced form still
|
||||
resolves for anyone installing holocron natively, so skill bodies written for both audiences
|
||||
should name the bare skill.
|
||||
|
||||
**Correction (2026-08-17): the evidence under the correction above is gone, and the claim goes with
|
||||
it — not to its opposite.** Observed on this machine: `~/.claude/plugins/installed_plugins.json` is
|
||||
`{"version": 2, "plugins": {}}`; there is no `enabledPlugins` key anywhere in `~/.claude.json`
|
||||
(`grep -c enabledPlugins` returns 0); `~/.apm/marketplaces.json` is `{"marketplaces": []}`. The
|
||||
`holocron` entry in `~/.claude/plugins/known_marketplaces.json` survives, but a registered
|
||||
marketplace is not an installed plugin. So the user-scope installs the 2026-08-14 correction cited
|
||||
are not there, and neither is the state the *original* consequence described before it. The claim
|
||||
about the namespaced form has now been written twice off two different observations of the same
|
||||
machine, and this ADR has already reversed itself once on it. That is the finding: the fact is
|
||||
machine state, not a property of this decision, and it changes without any commit. No instruction
|
||||
file — `AGENTS.md`, `CONTEXT.md`, or a skill body — should assert either way whether
|
||||
`<plugin>:<skill>` resolves. The rule that survives every observation is the one that was always the
|
||||
actionable half: write the bare name, because it is the only form `apm install` produces.
|
||||
|
||||
**apm owns `.claude/settings.json`.** (ADR-0019 supersedes the "exactly `{"hooks": {}}`" claim
|
||||
below — once a package ships a hook, apm merges it into that file and the merged entry is apm's own
|
||||
output. The rule that nothing repo-authored goes in the file is unchanged.) `apm audit --ci` replays the install into a scratch tree and
|
||||
diffs it against the worktree. apm's hook integrator writes that file, so the replay expects
|
||||
exactly what apm would have written — `{"hooks": {}}` — and any repo-owned key in it is permanent
|
||||
drift that fails the `apm-audit-ci` pre-push hook. Verified both directions: with the pre-existing
|
||||
`enabledPlugins` block present, `1 of 10 check(s) failed`; reduced to `{"hooks": {}}`,
|
||||
`All 10 check(s) passed`. Nothing was lost in that reduction — `enabledPlugins` was empty after the
|
||||
native uninstall and the only `hooks` entry was an empty `PreToolUse: []` — but it does mean the
|
||||
file is no longer available for repo-owned settings. Machine-specific settings go in the gitignored
|
||||
`.claude/settings.local.json`, which apm does not deploy; shared enforcement belongs in
|
||||
`.pre-commit-config.yaml`, where this repo already keeps it.
|
||||
|
||||
**`apm_modules/` breaks naive tree walks.** apm materializes a full copy of every dependency there
|
||||
— including this repo's own plugins, `.bats` files and all. The dependency copies resolve their
|
||||
bats helpers relative to their own root, not this repo's, so `tests/run-tests.sh` went from 167
|
||||
tests passing to `334 tests, 167 failures` on the first install. Both discovery walks
|
||||
(`tests/run-bats.sh`, `tests/run-tests.sh`) now exclude `apm_modules/`, on the find side and on the
|
||||
`git ls-files` side that derives the expected set. Any future script that walks the repo tree needs
|
||||
the same exclusion.
|
||||
|
||||
**Install output is gitignored; the lockfile is not.** `.claude/skills/`, `.claude/agents/`, and
|
||||
`apm_modules/` are regenerated by `apm install`. Committing the deployed skills would add a third
|
||||
mirror of content ADR-0017 already governs two copies of. `apm.lock.yaml` is committed — it is what
|
||||
makes the install reproducible, and `apm audit --ci` checks it.
|
||||
|
||||
**MCP survived the switch; hooks were never at risk.** apm read `plugins/bin/.mcp.json` as a
|
||||
self-defined direct-dependency MCP server and configured `obsidian` into the repo's `.mcp.json`
|
||||
unprompted. The `gitea` and `context7` servers were never plugin-provided — they live in
|
||||
`~/.claude.json` and are untouched. Every plugin's `.apm/hooks/hooks.json` is `{"hooks": {}}`, so
|
||||
apm's "contributed no entries to claude settings; skipped" warning on `kyberforge` and `lint` is
|
||||
accurate and harmless.
|
||||
|
||||
**A `.apm/` edit now needs a round trip.** The dependency resolves from the remote, so an edit is
|
||||
invisible to the running session until it is pushed and the install is refreshed. Under the native
|
||||
install with `autoUpdate` the shape was the same; it was more noticeable here at first because the
|
||||
refresh is a manual step where marketplace auto-update was not — ADR-0019 automates it at
|
||||
`SessionStart`.
|
||||
|
||||
**Correction (2026-08-14): the refresh command is `apm update`, not `apm install`.** An earlier
|
||||
revision of this paragraph named `apm install`, which is wrong: `apm install` deploys from the
|
||||
pinned `resolved_commit` in `apm.lock.yaml` and does not re-resolve refs (`apm install --force`
|
||||
documents this explicitly — "does NOT refresh refs; use 'apm update' for that"). Running it after a
|
||||
merge redeploys the same content and reports success.
|
||||
|
||||
**User scope is untouched, deliberately.** This decision changed project scope only; whatever is
|
||||
natively installed at user scope was left alone, and converting it is a separate decision with a
|
||||
blast radius beyond this repo. The specific inventory this paragraph used to name
|
||||
(`bin@holocron`, `gitea@holocron`, a stale `hello-world@holocron`) is machine state and is stale —
|
||||
see the 2026-08-17 correction above. The decision recorded here is unaffected by what that state is.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
- **`apm install --global`.** Verified working in an isolated `HOME`: user-scope deploys land in
|
||||
`~/.claude/skills/` and `~/.claude/agents/`, and it is the only scope where a plugin's `bin/`
|
||||
executables deploy (moot here — every `bin/` in this repo is empty but for a README). Deferred,
|
||||
not rejected: it changes skill resolution for every project on the machine at once.
|
||||
- **Keeping both install paths.** Rejected: the same skill would be present twice under two names,
|
||||
and `.claude/settings.json` cannot hold `enabledPlugins` without failing `apm audit --ci`.
|
||||
@@ -0,0 +1,167 @@
|
||||
# A SessionStart hook keeps the apm install current, replacing a git hook that never ran
|
||||
|
||||
ADR-0018 switched this repo to consuming its own plugins through `apm install`, with the six
|
||||
packages declared as unpinned git refs against the holocron remote's default branch. That decision
|
||||
left a hole it named but did not fill: the deployed content goes stale the moment anyone merges,
|
||||
and nothing detects it.
|
||||
|
||||
**Status: accepted (2026-08-14).**
|
||||
|
||||
## Context
|
||||
|
||||
The pre-existing answer was `scripts/git-hooks/post-push`, which pulled the marketplace clone and
|
||||
ran `claude plugin update kyberforge`. Issue #78 filed it as a bug — the hook updated `kyberforge`
|
||||
but not `gitea`, so gitea skills stayed pinned at a pre-refactor version after #67 merged.
|
||||
|
||||
The issue's premise was wrong in a way nobody had noticed for six weeks. **Git has no client-side
|
||||
`post-push` hook.** `githooks(5)` does not list one, and git 2.39.5 does not invoke one.
|
||||
`scripts/install.sh` copies every file in `scripts/git-hooks/` into `.git/hooks/`, so
|
||||
`.git/hooks/post-push` existed on disk and looked installed. It had never fired. The hook did not
|
||||
skip `gitea`; it skipped everything. Both tests that appeared to cover it — `test-post-push.sh` and
|
||||
`test-git-hooks-install.sh` — asserted only that the script behaved correctly when invoked directly
|
||||
and that install.sh copied the file. Neither asserted that git ever runs it.
|
||||
|
||||
That also makes the original framing wrong. Refreshing on push assumes the person who pushes is the
|
||||
person who goes stale, which is backwards: your install goes stale when *someone else* merges, and a
|
||||
push of your own is neither necessary nor sufficient for it to have happened.
|
||||
|
||||
## Decision
|
||||
|
||||
A `SessionStart` hook, shipped in `plugins/kyberforge/.apm/hooks/`, checks whether the install is
|
||||
behind and refreshes it in place.
|
||||
|
||||
`SessionStart` is the correct trigger because the thing that goes stale is the skill content a
|
||||
*session* loads, and that is the moment the staleness does damage. It also enables two things a git
|
||||
hook structurally cannot do: `additionalContext` puts the notice into the agent's context rather
|
||||
than terminal scrollback nobody reads, and `reloadSkills: true` makes the host re-scan the skill
|
||||
directories after the hook returns, so a refresh lands in the running session without a restart.
|
||||
|
||||
apm's own lifecycle events (`pre-/post-install`, `pre-/post-update`, `pre-/post-uninstall`) were
|
||||
rejected: they fire around apm operations already chosen, so they can announce a refresh but never
|
||||
detect that one is needed.
|
||||
|
||||
Three sub-decisions:
|
||||
|
||||
- **Refresh automatically rather than report.** The hook runs `apm update --yes` and asks for a skill
|
||||
reload. The rejected alternative was to report and let a human run it. Auto-refresh costs a
|
||||
rewritten `apm.lock.yaml` — a committed file — appearing as an unexplained modification in the
|
||||
working tree, on any branch, at any time. The emitted notice says so explicitly for that reason.
|
||||
- **`plugins/kyberforge/.apm/hooks/`, not `.claude/settings.json`.** ADR-0018 established that apm
|
||||
owns `.claude/settings.json` and that any repo-authored key in it is permanent `apm audit --ci`
|
||||
drift. A hook shipped in a package is written into that file by apm itself, so it is apm's output
|
||||
and does not drift. `.claude/settings.local.json` also works but is gitignored and machine-local,
|
||||
which fails the requirement that this travel with the repo.
|
||||
- **`startup` matcher only.** `resume`, `clear`, `compact` and `fork` would re-run the check on every
|
||||
compaction, and a compaction is not an event after which the remote can have moved.
|
||||
|
||||
The executable-trust gate is switched on at the same time. Root `apm.yml` gains an `executables:`
|
||||
block allowing kyberforge's hooks and bin.
|
||||
|
||||
## Consequences
|
||||
|
||||
**The gate is off until something turns it on, and this repo had it off.** `apm approve --list`
|
||||
reports `Executable-trust gate disabled -- all executables deploy` until an `executables:` block
|
||||
exists in `apm.yml`. Any hook, bin, or MCP primitive a dependency shipped would have deployed with
|
||||
no prompt and no record. The block added here closes that for this repo; every other apm project on
|
||||
this machine still has it open.
|
||||
|
||||
**The allow key is version-pinned, and that is a live failure mode.** apm writes
|
||||
`kyberforge#1.5.0`, not `kyberforge` — and the release that ships this hook proved the point
|
||||
immediately, since bumping kyberforge to 1.5.0 required editing the key in the same commit. A
|
||||
kyberforge version bump makes the entry stop matching, the
|
||||
gate blocks the hook, and the install silently stops refreshing — the exact failure this ADR exists
|
||||
to end, reintroduced through the mechanism meant to secure it.
|
||||
|
||||
Matching is an exact dictionary lookup on the composed `name#version` string
|
||||
(`apm_cli/security/executables.py`, `is_package_approved`), so there is no wildcard or
|
||||
version-less key that would sidestep this — the key has to be edited on every bump, and the
|
||||
question is only what catches a missed edit. A comment in the `executables:` block is not enough:
|
||||
this repo gates generated-content drift, marketplace mirror drift and vale style drift
|
||||
deterministically, and a silent-staleness failure is strictly worse than any of them. So
|
||||
`scripts/check-executables-allow-sync.sh` runs at pre-push, parsing `version:` out of
|
||||
`plugins/kyberforge/apm.yml` and asserting root `apm.yml` carries the matching
|
||||
`kyberforge#<version>` key. The comment stays as the human-facing pointer; the hook is what
|
||||
actually holds. It parses with PyYAML where importable and falls back to a two-shape scan
|
||||
otherwise, so a missing pip package cannot become the thing that blocks every push.
|
||||
|
||||
**Trust is keyed on the version, not on the content.** `kyberforge#1.5.0` approves whatever
|
||||
`check-apm-current.sh` contains at the moment it is fetched, not the bytes that were reviewed when
|
||||
the key was written. Because the dependency is unpinned against the default branch and the hook
|
||||
runs `apm update --yes` unattended, an edit to that script landing on `main` deploys and executes
|
||||
on every contributor's machine at their next session start, with no second approval prompt and no
|
||||
diff shown. The trust gate constrains *which package* may ship an executable; it does not constrain
|
||||
what that executable does between version bumps. That is an accepted property of this design rather
|
||||
than an oversight — the remote is self-hosted, push access to `main` is already sufficient to
|
||||
change any skill body an agent will follow — but it is the reason the gate should not be read as a
|
||||
supply-chain control. Pinning each dependency to a `ref:` is what would make it one, and ADR-0018
|
||||
defers that until per-package release tags exist.
|
||||
|
||||
**A referenced hook script must be addressed at its `.apm/` path.** apm resolves
|
||||
`${CLAUDE_PLUGIN_ROOT}/...` against the installed package root, and `apm pack` keeps only `*.json`
|
||||
from `.apm/hooks/` when it builds the flat mirror. So `${CLAUDE_PLUGIN_ROOT}/hooks/check-apm-current.sh`
|
||||
resolves to the mirror, where the script does not exist — verified, apm reports
|
||||
`Hook script not found` and deploys a hook pointing at nothing. The working reference is
|
||||
`${CLAUDE_PLUGIN_ROOT}/.apm/hooks/check-apm-current.sh`. The script cannot simply be placed in
|
||||
`plugins/kyberforge/hooks/` either: that directory is `rm -rf`'d by every content sync (ADR-0017).
|
||||
A test pins the reference.
|
||||
|
||||
**Session startup gets slower when the install is stale.** Measured: ~0.7 s for the `apm outdated`
|
||||
check when everything is current, ~10.4 s when six packages are behind and the refresh runs. The
|
||||
hook declares `timeout: 380` to cover a cold multi-package fetch. That number is not free-standing:
|
||||
the script imposes its own `timeout 60` on `apm outdated` and `timeout 300` on `apm update`, so the
|
||||
host-side timeout has to exceed their sum or the host kills the hook mid-update and leaves
|
||||
`.claude/skills/` half-deployed with no notice emitted. An earlier revision declared `320`, which
|
||||
was below the 360 s the script can legitimately take. A test asserts the invariant rather than the
|
||||
literal — it parses every `timeout N` out of the script, sums them, and requires the `hooks.json`
|
||||
value to be larger — so changing either side without the other fails the suite.
|
||||
|
||||
**Reading a human-readable CLI for a control decision cost a silent failure, again.** `apm outdated`
|
||||
has no `--json` or other machine-readable flag (confirmed against 0.28.0), so the hook must match
|
||||
its prose. The first attempt matched `outdated dependencies found` — plural only. apm emits
|
||||
`1 outdated dependency found` in the singular when exactly one package is behind
|
||||
(`apm_cli/commands/outdated.py`), so a single stale package was invisible: the hook exited 0
|
||||
silently and no refresh ran. With six packages merging independently, one-behind is the ordinary
|
||||
case rather than an edge, which means the mechanism failed most often in exactly the situation it
|
||||
exists for. The match is now `outdated dependenc(y|ies) found`.
|
||||
|
||||
The deeper lesson is the one `post-push` already taught and this repeated: every assertion about the
|
||||
hook mocked `apm`, so the suite was green while the hook could not detect the common case. Mocks
|
||||
verify the code against its author's belief about the interface, never the interface. The suite now
|
||||
carries one probe that stages a genuinely outdated dependency against a local git remote — offline,
|
||||
via `url.<path>.insteadOf`, so the twelve-hooks-pass-under-`unshare -rn` property survives — runs
|
||||
the real `apm outdated`, and replays its genuine output through the real hook. Reverting the grep
|
||||
to plural-only fails it.
|
||||
|
||||
**The hook cannot install itself.** Dependencies resolve from the remote, so the hook does not
|
||||
deploy until this change is merged and `apm update` has run once against the new default branch.
|
||||
Until then the repo has the mechanism in source and not in effect.
|
||||
|
||||
**`.claude/settings.json` stops being `{"hooks": {}}`.** apm merges the hook into it and tracks
|
||||
ownership in a `.claude/apm-hooks.json` sidecar, with the script copied to
|
||||
`.claude/hooks/<pkg>/`. The sidecar and the script directory are gitignored install output; the
|
||||
settings file remains committed, now with apm-generated content in it. ADR-0018's statement that the
|
||||
committed content is exactly `{"hooks": {}}` is superseded on that point only — the rule it was
|
||||
protecting, that nothing repo-authored goes in that file, is unchanged.
|
||||
|
||||
**Native consumers are protected by a guard, not by the gate.** A host installing holocron through
|
||||
`claude plugin install` auto-discovers `hooks/hooks.json` and does not consult apm's trust gate at
|
||||
all. The script therefore exits silently when there is no `apm.lock.yaml` in the working directory,
|
||||
which is what makes it inert in a repo that does not consume packages through apm. Copilot CLI sees
|
||||
no hook at all, for the reasons already documented in `plugins/kyberforge/docs/hooks.md`.
|
||||
|
||||
**`scripts/git-hooks/` is now empty.** `post-push` and `test-post-push.sh` are deleted.
|
||||
`install.sh`'s copy block is generic and is kept; `test-git-hooks-install.sh` now synthesizes its
|
||||
own fixture hook instead of depending on a real one existing, so the mechanism stays tested and can
|
||||
be used again if a hook git actually invokes is ever wanted.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
- **A `post-merge` git hook.** Real, unlike `post-push`, and verified to fire on both a
|
||||
fast-forward `git pull` and a `git pull --rebase`. Rejected as the primary mechanism because a
|
||||
pull is the wrong signal, and because it cannot reload skills in a running session. It remains
|
||||
the only option for a project that consumes apm packages without a Claude-family host.
|
||||
- **Reporting instead of refreshing.** See the sub-decision above.
|
||||
- **A seventh plugin holding only this hook**, to avoid shipping it to external kyberforge
|
||||
consumers. Rejected as disproportionate: the `apm.lock.yaml` guard already makes the hook inert
|
||||
for anyone not consuming through apm, and a package exists to be maintained, versioned, and
|
||||
registered in the marketplace.
|
||||
419
docs/adr/0020-skill-description-and-body-context-contract.md
Normal file
419
docs/adr/0020-skill-description-and-body-context-contract.md
Normal file
@@ -0,0 +1,419 @@
|
||||
# Skills and agents are authored against a context budget, not a spec ceiling
|
||||
|
||||
Every installed skill's `name` and `description` sits in every agent's context from the first token
|
||||
of every session, whether or not the skill is ever invoked. Across this repo's 39 skills that is
|
||||
23,427 characters — roughly 5,900 tokens — and the authoring rules that produced it optimised for
|
||||
triggering reliability with no counter-pressure on size. This ADR sets the budget, the shape, and the
|
||||
gates that hold them.
|
||||
|
||||
**Status: accepted (2026-08-14).**
|
||||
|
||||
## Context
|
||||
|
||||
Every `file:line` citation in this ADR is against the base commit the decision was taken on,
|
||||
`f9b919d7e3bd5e6b51fbdf88b32ace0438b313e0`, not against current `HEAD`. The change that carries this
|
||||
ADR rewrites several of the cited files, so a citation resolved against the worktree will land on
|
||||
unrelated text. Use `git show f9b919d:<path>` to follow one.
|
||||
|
||||
Measured before any change, at that commit. Method, so the figures are reproducible: sum
|
||||
`len(name) + len(description)` over the frontmatter of every `plugins/*/.apm/skills/*/SKILL.md`,
|
||||
folding `>` block scalars to the value the host actually loads (most descriptions here are folded
|
||||
scalars, so counting raw lines measures indentation instead); tokens at the standard
|
||||
~4-characters-per-token approximation `scripts/skill-size-check.sh` uses. Word counts are
|
||||
whitespace-separated tokens, and are stated as **body-only** or **whole-file** every time, never bare.
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| 39 skill `name` + `description` | 23,427 chars, ~5,900 tokens, **preloaded every session** |
|
||||
| 4 agent `name` + `description` | 1,325 chars, ~330 tokens, preloaded every session |
|
||||
| skill bodies (body-only words) | median 684, mean 815, p90 1,349 |
|
||||
| skill files (whole-file words) | median 816, mean 927, p90 1,526 |
|
||||
| `MAX_WORDS` gate (`skill-audit/scripts/validate.sh:147`) | **2,770** whole-file — a density proxy, not a percentile |
|
||||
|
||||
That last row is worth stating plainly, because it is the first thing this ADR is about. 2,770 is not
|
||||
derived from the corpus distribution at all: per the derivation comment in
|
||||
`scripts/skill-size-check.sh`, it is 2,770 words at the densest observed 7.22 chars/word ≈ 20,000
|
||||
chars ≈ the agentskills.io ~5,000-token ceiling. Neither percentile reaches it — 2× the body-only p90
|
||||
is 2,698 and 2× the whole-file p90 is 3,052 — and reading it as "2× p90" would pair a whole-file gate
|
||||
against a body-only distribution, which is exactly the conflation this ADR exists to stop.
|
||||
|
||||
Three findings drove this, none of which is "the descriptions drifted".
|
||||
|
||||
**The rules mandate the bloat.** `skill-author/SKILL.md:104` requires indirect triggers ("even if the
|
||||
user doesn't mention X explicitly") and `skill-audit/references/description-quality.md:21` requires
|
||||
authors to "err toward being pushy". Both are enforced. The one rule that would delete the waste —
|
||||
`skill-author/SKILL.md:102`, "not the skill's internal mechanics" — is judgment-only and is absent
|
||||
from the FAIL conditions at `description-quality.md:45-50`. The enforced rules inflate; the deflating
|
||||
rule does not bite. The result is measurable: `gitea-files` spends 147 chars listing six verbs, then
|
||||
301 chars re-quoting the same six as user phrasings, in the same order. `apm-workflow` does the same
|
||||
with six capability clusters. Across the twelve longest descriptions, 30.7% is capability
|
||||
enumeration and 11.6% is composition or implementation detail that cannot affect a routing decision.
|
||||
|
||||
**Capability enumeration in a description is a correctness hazard, not only a token cost.**
|
||||
`plugins/kyberforge/docs/research/examples/skill-write/writing-skills/SKILL.md:154-158` reports a
|
||||
measured failure: "when
|
||||
a description summarizes the skill's workflow, an agent may follow the description instead of reading
|
||||
the full skill content. A description saying 'code review between tasks' caused an agent to do ONE
|
||||
review, even though the skill's flowchart clearly showed TWO reviews." `git-commits` is exactly that
|
||||
shape — 74% of its description is capability enumeration, including a rules table (`header max 100
|
||||
chars, lowercase subject, no trailing periods, 11 standard types`) an agent can act on without ever
|
||||
loading the body.
|
||||
|
||||
**The upstream sources cannot settle this.** The four skill-writing references under
|
||||
`plugins/kyberforge/docs/research/examples/skill-write/` disagree on what a description contains —
|
||||
when-only (`writing-skills/SKILL.md:99`), what-and-when (`skill-creator/SKILL.md:67`,
|
||||
`writing-skills/anthropic-best-practices.md:187`), triggers-only
|
||||
(`writing-great-skills/SKILL.md:28`), and what-plus-when-plus-negative
|
||||
(`write-skill/SKILL-TEMPLATE.md:5-6`). Those four paths are relative to that directory.
|
||||
`writing-skills` and the Anthropic document it bundles contradict each other inside one skill
|
||||
directory. They also disagree on whether
|
||||
500 lines is binding, on the inline-versus-bundle threshold, and on the TOC threshold (>100 lines vs
|
||||
>300 lines). "Grounded in the research" is therefore not available as a tiebreaker; a house choice is
|
||||
required and this is it.
|
||||
|
||||
A fourth observation shaped the body half. The best progressive-disclosure ratio in the repo belongs
|
||||
to `apm-workflow` — a 421-word body dispatching to 3,006 words of references — and the worst two
|
||||
belong to the skills that define the house standard: `skill-author` (2,623-word body / 1,247 words of
|
||||
references) and `agent-author` (2,582 / 1,664). Measured the other way, whole-file, those two are
|
||||
2,760 and 2,758 words — ten and twelve words under the 2,770 gate their own plugin enforces. A
|
||||
ceiling that nothing approaches is not a constraint; a ceiling that two files have grown into is a
|
||||
target. The two numbers for one file are the point: 2,623 and 2,760 describe the same `skill-author`,
|
||||
and only one of them is what either gate measures.
|
||||
|
||||
## Decision
|
||||
|
||||
### Descriptions
|
||||
|
||||
A description carries three things and nothing else: a **trigger clause**, at most one **capability
|
||||
clause**, and a **boundary clause**. Capability enumeration, output-format detail, composition notes
|
||||
("composes X rather than duplicating Y"), and implementation detail move to the body or to
|
||||
`README.md`.
|
||||
|
||||
- **250 characters SUGGESTION, 400 FAIL.** The agentskills.io 1,024-character limit remains as an
|
||||
unchanged spec backstop. The SUGGESTION tier is what moves the average; the FAIL tier only stops
|
||||
outliers.
|
||||
- **A missing, valueless or `null` `description:` is a hard FAIL** in all three validators. That
|
||||
reads as a trivial precondition and is not: a `description:` line with no value followed by
|
||||
`model: sonnet` let a line regex capture the *next* key, which looked non-empty, so the "missing or
|
||||
empty" branch never fired and every gate below it then early-returned on the genuinely empty folded
|
||||
value — exit 0, zero output, on a blocking pre-push gate. Presence is decided on the YAML-folded
|
||||
value and nowhere else. The field this contract is entirely about is the one field a gate must
|
||||
never fail to notice is absent.
|
||||
- **Boundary clauses compress** to `Not <thing> → <skill-name>.` and must name a target that
|
||||
resolves to a real skill or agent. Resolution walks up **from the file being checked** to an
|
||||
*authoring root* — the nearest ancestor holding `plugins/*/.apm/skills` or `plugins/*/.apm/agents`,
|
||||
falling back to the nearest ancestor holding `.git`. Two passes rather than one interleaved walk,
|
||||
so a nested `.git` (a submodule, a sub-package worktree) cannot beat a real monorepo root further
|
||||
up. When an authoring root is found the universe is every skill and agent under
|
||||
`<root>/plugins/*/`, plus the target's own apm package and the packages that package declares in
|
||||
its own `apm.yml` `dependencies.apm`. Sibling plugins resolve against each other, which is what a
|
||||
monorepo means. Deployed `.claude/`/`.agents/` trees are consulted **only** when the walk found no
|
||||
plugin monorepo root — whether it landed on a bare `.git` ancestor or on nothing at all. That is
|
||||
the consumer case, where there is no monorepo to read. The condition is which of the two passes
|
||||
matched, never a name-count delta: a single-plugin monorepo re-collects its own package and adds
|
||||
no new name, so a delta test reads zero there and would pull the deployed trees back in. What the
|
||||
resolver must never do is
|
||||
derive the universe from its own location: a `${BASH_SOURCE}`-relative repo root leaked this repo's
|
||||
39-skill universe into every consumer repo running the hook through pre-commit, so a consumer skill
|
||||
routing to `skill-audit` resolved against a plugin it had never installed. Checked
|
||||
deterministically. A description carrying **no** boundary clause at all is a SUGGESTION, for skills
|
||||
and agents alike: most descriptions want one, some genuinely have no near-miss sibling to exclude,
|
||||
and that judgment is not a script's to make.
|
||||
- **The verdict must not depend on whether `apm install` has been run.** Deployed trees are
|
||||
gitignored install output, present only on a machine that has run it. Four cross-plugin targets
|
||||
here (`gitea-branches` → `git-branches`, `gitea-branches` → `git-history`, `gitea-issues` →
|
||||
`git-branches`, `gitea-workflow` → `git-workflow`) once resolved through `.claude/skills/` alone,
|
||||
so the same commit measured 2 dangling targets on a developer machine and 6 on a fresh clone. A
|
||||
gate shipping hot with no baseline cannot give two answers. Under the walk-up those four resolve
|
||||
because sibling plugins are in the universe — no plugin here declares a cross-plugin apm
|
||||
dependency, and none needs to. Verified: a tree holding only `plugins/` and the root `apm.yml`,
|
||||
with no `.claude/` or `.agents/` anywhere, now produces findings identical to the working tree —
|
||||
26 description FAILs, 9 body FAILs, 2 dangling targets, 0 missing references, 58 SUGGESTIONs.
|
||||
- **The universe is the apm marketplace, and nothing else.** A routing target resolves to a skill or
|
||||
an agent, or it does not resolve. Host built-ins are deliberately outside it: `/compact`, `/clear`
|
||||
and `/init` are Claude Code slash commands with no counterpart in Copilot CLI or Codex, so a
|
||||
vendor-neutral `.apm/` description routing to one is a portability defect and the hard FAIL is a
|
||||
true positive, not a false one. An allowlist of known built-ins was **rejected**: it answers a
|
||||
different question ("does this exist on *some* host?"), it cannot answer that portably from a
|
||||
single source file, and it goes stale the next time a host ships a command — reintroducing the
|
||||
same-commit-two-verdicts failure the bullet above exists to close. An author who needs to mention
|
||||
one writes it un-slashed (``the `compact` built-in``), which is not route notation and makes no
|
||||
routing claim.
|
||||
- **Blocking is scoped to a sentence, which makes sentence boundaries load-bearing.** A prose-form
|
||||
target earns a hard error only when its own sentence names another target that *resolves*; route
|
||||
notation (`/name`, `→ name`) is exempt and always blocks. So the splitter is part of the contract,
|
||||
not a detail of it. `e.g. "…"` is not a sentence end, and a sentence opening with a code span or a
|
||||
lowercase skill name is a start; getting either wrong moves targets between the two tiers in
|
||||
opposite directions — a stranded corroborator silently demotes a real finding to SUGGESTION, and a
|
||||
missed boundary lets one sentence vouch for a target it never stood beside, producing a hard FAIL
|
||||
with no escape hatch.
|
||||
- **The blanket pushiness rules are deleted.** `skill-author/SKILL.md:104` and
|
||||
`description-quality.md:21` are replaced by a conditional: add an indirect trigger only where the
|
||||
user's natural phrasing genuinely omits the domain word — true for the `gitea-*` family, false for
|
||||
`git-commits`. Stating the same trigger twice in two registers is a FAIL.
|
||||
|
||||
### Bodies
|
||||
|
||||
The body carries the **decision procedure only**: ordered steps, decision branches, gates, and which
|
||||
reference to load when. Lookup tables, spec restatements, output schemas, templates, and rationale
|
||||
prose move to `references/` behind an explicit "read X when Y" trigger.
|
||||
|
||||
- **600 words SUGGESTION, 900 FAIL, counted body-only** — everything after the closing `---` of the
|
||||
frontmatter. The 2,770-word / 500-line spec backstop is unchanged, keeps its existing meaning
|
||||
(conformance, not quality), and keeps counting the **whole file including frontmatter**. These are
|
||||
two different gates measuring two different things, and conflating them is what produced the
|
||||
current state.
|
||||
- **Dispatch is mandatory at two or more mutually exclusive flows.** The body carries the dispatch
|
||||
table and the gates that apply to every branch; each flow lives in its own self-contained
|
||||
`references/` file. This is `apm-workflow/SKILL.md:33-41` promoted from accident to rule. "Two
|
||||
mutually exclusive flows" is not decidable from file text, so this rule is auditor judgment — see
|
||||
Enforcement below for what that means and does not mean.
|
||||
- **Every `references/<file>.md` a body names must exist.** A dispatch table pointing at a file that
|
||||
was never written is a silently dead branch. Checked deterministically.
|
||||
- **Gotchas are constrained.** A Gotcha must state a fact that contradicts a reasonable default —
|
||||
something the agent gets wrong by acting sensibly. More than five entries is a SUGGESTION, as is a
|
||||
Gotchas section exceeding 25% of the body; both are countable and both are checked
|
||||
deterministically. A Gotcha that paraphrases a step in the body below it is a FAIL, but a FAIL an
|
||||
auditor issues, not a script — semantic equivalence is not pattern-matchable.
|
||||
|
||||
### Agents
|
||||
|
||||
Agents take the same description gates — they are preloaded identically — and **no body word gate**.
|
||||
A skill body is loaded into the caller's context, competing with the live conversation; an agent body
|
||||
becomes the system prompt of a fresh context. The rationale for the 900-word FAIL does not transfer.
|
||||
|
||||
That exemption is expressed in `agent-audit/scripts/validate.sh`, which has no body constant, and in
|
||||
the `files:` pattern of the `skill-size-check` pre-commit hook, which is `SKILL.md`-only. It is *not*
|
||||
expressed in `scripts/skill-size-check.sh` itself, which measures whatever path it is handed —
|
||||
running it directly over `plugins/*/.apm/agents/*.agent.md` today reports 900-word body FAILs on
|
||||
`git-orchestrate` (933), `gitea-orchestrate` (1,199) and `apm-orchestrate` (1,080). Agents escape by
|
||||
file pattern, not by the script knowing the difference. Anyone widening that pattern to cover agents
|
||||
would silently enforce a gate this ADR declines to set.
|
||||
|
||||
A plugin-scope agent is a single file with no sibling `references/` directory, so it cannot disclose
|
||||
to itself — it can only delegate to skills. `agent-audit` therefore gains a **delegation check**: an
|
||||
agent body that restates a procedure owned by a skill it can invoke is a FAIL, with the fix being
|
||||
"invoke `<skill>` instead". Length falls out of delegation rather than being gated directly.
|
||||
|
||||
### Invocation as a design axis
|
||||
|
||||
`skill-author` asks whether a skill is model-invoked or hand-invoked before writing a description. A
|
||||
hand-invoked skill sets `disable-model-invocation: true` and carries one plain human-facing sentence
|
||||
with no trigger list.
|
||||
|
||||
Verified end-to-end rather than assumed: `plugins/bin/.apm/skills/zoom-out/SKILL.md:4` carries the
|
||||
flag, apm passes it through verbatim to both `.claude/skills/zoom-out/SKILL.md:4` and the flat mirror
|
||||
at `plugins/bin/skills/zoom-out/SKILL.md:4`, and `zoom-out` is the one installed skill absent from
|
||||
the model-visible skill listing in a live session. It remains invocable as `/zoom-out`.
|
||||
|
||||
### Merging siblings
|
||||
|
||||
Two skills that share substantial content, name each other as near-misses, and differ only in the
|
||||
type of input they take should be **one skill with a dispatch table**. This catches `skill-audit` +
|
||||
`agent-audit` and is scoped to them; the author pair is explicitly excluded, because
|
||||
`skill-author` and `agent-author` emit genuinely different artifacts (a skill directory versus a
|
||||
one-or-two-file agent pair, per ADR-0005 and ADR-0016) and their overlap is in the improve flow
|
||||
rather than the core job.
|
||||
|
||||
**DEFERRED — not implemented in the change that carries this ADR. Tracked as issue #101.** Both
|
||||
skills still exist separately, and this change made the split deeper rather than shallower: retrofit
|
||||
to the dispatch pattern took `skill-audit` from 3 reference files to 7 and `agent-audit` from 4 to 8,
|
||||
and their two same-named `references/description-quality.md` files now differ on 100 of ~120 lines
|
||||
after normalising `skill`/`agent`, where before they were closer. The merge stays the decision; it
|
||||
reopens ADR-0008 (agent-audit's single-file invocation contract) and touches every call site in
|
||||
`skill-author`, `agent-author` and `forge`, which is why it is its own change and not a rider on
|
||||
this one. Recorded here rather than dropped, so the gap between the rule and the tree is deliberate
|
||||
and dated instead of discovered later.
|
||||
|
||||
### Enforcement and rollout
|
||||
|
||||
Gates land where the existing gates already live — no new layer. The table below is exhaustive about
|
||||
which tier each rule is in, because the failure this ADR is most exposed to is a rule filed under
|
||||
"Enforcement" that no validator implements:
|
||||
|
||||
| Check | Applies to | Tier | Home |
|
||||
|---|---|---|---|
|
||||
| description characters (250 SUGGESTION / 400 FAIL) | skills, agents | deterministic | `scripts/skill-size-check.sh`; constants mirrored in `skill-audit/scripts/validate.sh` and `agent-audit/scripts/validate.sh` |
|
||||
| body-only words (600 SUGGESTION / 900 FAIL) | skills | deterministic | `skill-size-check.sh`, `skill-audit/scripts/validate.sh` |
|
||||
| description present and non-empty (ERROR) | skills, agents | deterministic | same |
|
||||
| boundary target resolves to a real skill or agent (ERROR when written as `/name` or `-> name`, or when its own sentence names another target that resolves; SUGGESTION otherwise) | skills, agents | deterministic | same |
|
||||
| boundary clause absent (SUGGESTION) | skills, agents | deterministic | same |
|
||||
| Gotchas entry count over five (SUGGESTION) | skills | deterministic | same |
|
||||
| Gotchas over 25% of the body (SUGGESTION) | skills | deterministic | same |
|
||||
| every `references/<file>.md` a body names exists (ERROR) | skills | deterministic | same |
|
||||
| description opener, composition notes in a description | skills, agents | prose pattern | `plugins/kyberforge/.apm/skills/*/assets/vale/styles/Kyberforge/` |
|
||||
| a Gotcha paraphrasing a body step | skills | **auditor judgment** | `references/body-discipline.md` |
|
||||
| dispatch at two or more mutually exclusive flows | skills | **auditor judgment** | `references/body-discipline.md` |
|
||||
| delegation: an agent body restating a skill's procedure | agents | **auditor judgment** | `agent-audit` |
|
||||
| capability enumeration, restatement, trigger quality | skills, agents | **auditor judgment** | `references/description-quality.md` |
|
||||
|
||||
The rows in bold are stated as FAILs in the Decision above and are FAILs an *auditor* issues. None of
|
||||
them is countable: "does this Gotcha paraphrase step 4", "are these two flows mutually exclusive" and
|
||||
"does this agent body restate what `git-commits` already owns" are semantic questions, and a script
|
||||
that guessed at them would be a worse gate than no gate, because it would be believed. They are not
|
||||
enforced, they are reviewed, and this table exists so that distinction is written down rather than
|
||||
inferred from whether a validator happens to have been written yet.
|
||||
|
||||
Two of the deterministic rows are tuned for **false positives over recall**, and what they decline to
|
||||
see is part of the contract. On target extraction: a bare hyphenated name counts only inside a
|
||||
boundary sentence, and a single-word name is never matchable bare — `research`, `triage`, `forge`,
|
||||
`prototype` and `tdd` are all real skill names *and* ordinary English, so it must be written
|
||||
`` `forge` `` or `/forge` to be seen at all. Grammar then decides whether a recognised target may
|
||||
raise an error: one followed by an ordinary lowercase noun is a compound **modifier**, not a route
|
||||
("use pre-commit hooks instead of ad-hoc scripts", "invoke the pull-request template"), so it is
|
||||
confirm-only — it still resolves and still counts as a route when the name exists, but it can never
|
||||
dangle. Only a *terminal* target can. The compressed arrow form `→ <name>` is exempt from that
|
||||
follower test and is always error-eligible, because nothing reads as a compound modifier after an
|
||||
arrow; a `/slash` target reached through a route verb is **not** exempt and takes the same test. The
|
||||
simpler rule — "only marked targets may dangle" — was available and would have been wrong here: both
|
||||
live true positives are bare, `research`'s "(use neuledge-context)" and the `gitea-labels-` /
|
||||
`milestones` fold. On the body-shape checks: a `## Gotchas` heading must *end* in "gotchas", not
|
||||
merely contain the word, so `## Gotcha handling` and `## Why gotchas matter` are prose sections and
|
||||
are skipped; fenced code blocks are masked out of heading detection and entry counting, so a fenced
|
||||
example list is not mistaken for the section; and a `references/` pointer named on a line
|
||||
that also says the file is gone ("removed", "deprecated", "no longer") is read as a historical
|
||||
mention rather than a dead dispatch entry. Note the 25% fraction is deliberately *not* fence-masked
|
||||
on either side — fenced lines are real body words, and the fraction is measured against the whole
|
||||
body.
|
||||
|
||||
**The deterministic tier blocks immediately, with no baseline file.**
|
||||
|
||||
Three pre-existing contradictions are fixed in the same change, because they are the contract:
|
||||
|
||||
- `skill-audit/SKILL.md:58` asks whether the description opens with an action verb ("Audits…",
|
||||
"Reviews…"), while `:56` defers the same question to `Kyberforge.DescriptionOpener` and
|
||||
`skill-author/SKILL.md:101` requires an imperative "Use when…" opener. The criterion is
|
||||
unsatisfiable against the house's own skills, both of which open with "Use when".
|
||||
- `DescriptionOpener.yml` is anchored to `^This (skill|agent)\b`, which misses a plain `This …`
|
||||
opener; it is widened here to `^This\b`. The anchor itself stays. Composition prose that sits
|
||||
*mid*-description — `gitea-workflow`'s "This is the human-facing entry point…" at character 377,
|
||||
`gitea-labels-milestones`'s "This is a cross-cutting shared skill…" at character 300 — was never in
|
||||
the opener rule's scope and correctly is not: under `scope: text.frontmatter.description` the `^`
|
||||
anchors to the start of the whole folded value, and un-anchoring to reach mid-description text was
|
||||
measured at 5 hits and 5 false positives and rejected (`LESSONS.md`, 2026-08-14). The real gap is
|
||||
that no rule covered that text at all, which a new token-list rule, `Kyberforge.CompositionNote`,
|
||||
closes: 10 alerts across four `gitea-*` skills, 0 false positives.
|
||||
- `description-quality.md:45-50` has no FAIL condition for internal-mechanics content, which is why
|
||||
`skill-author/SKILL.md:102` never bit.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Editing any non-compliant skill now requires retrofitting it first.** At decision time, 30 of 39
|
||||
descriptions exceeded 400 characters and 13 of 39 bodies exceeded 900 words — the latter counted
|
||||
body-only, which is what the new gate measures; the pre-existing 2,770-word gate counts the whole
|
||||
file including frontmatter, and the two must not be conflated. The change that carries this ADR also
|
||||
retrofits kyberforge's own four author/audit skills, so the figures on landing are **26 and 9**.
|
||||
With the gate hot and no baseline, a one-line
|
||||
fix to `gitea-prs` cannot be committed until that skill meets the contract. This is deliberate — it
|
||||
guarantees convergence and avoids a half-state — but it means the retrofit is lazy and *mandatory*
|
||||
rather than deferred. Issue #99 tracks it and should be prioritised accordingly, and the risk it
|
||||
carries is the ordinary one for hot gates: a gate expensive enough to be inconvenient gets bypassed
|
||||
with `SKIP=` and loses its authority.
|
||||
|
||||
**A second hot gate ships alongside it, and it is easy to miss.** `Kyberforge.CompositionNote` is
|
||||
`level: error` like every other rule in that style, so `pre-commit run --all-files` is red on 10
|
||||
alerts across `gitea-issues`, `gitea-labels-milestones`, `gitea-prs` and `gitea-workflow`
|
||||
independently of anything `skill-size-check` reports. Someone scoping the #99 retrofit off the size
|
||||
findings alone will fix those and still be blocked. The two gates want fixing together.
|
||||
|
||||
**A ceiling does not produce an average.** If every author writes to the 400-character FAIL, the
|
||||
preload lands at 39 × 400 = 15,600 chars — a 33% cut off 23,427, not the ~50% intended. Writing to
|
||||
the 250-character SUGGESTION instead lands at 9,750, a 58% cut. The halving depends entirely on the
|
||||
250-character SUGGESTION tier being visible and respected. That tier works here in a way it does not
|
||||
elsewhere in this repo: `skill-audit` already reports `PASS (N suggestions)` as a first-class
|
||||
outcome. This is explicitly **not** the failure ADR-0013 records — Vale warnings are invisible
|
||||
because vale's exit code keys on `error` alone, but these gates live in `validate.sh` and
|
||||
`skill-audit`, where a SUGGESTION reaches the report. Realistic landing is somewhere in that 33-58%
|
||||
band, not a guaranteed 50%.
|
||||
|
||||
**A word gate cannot detect the defect it is standing in for.** `git-commits` carries twelve Gotchas
|
||||
of which four restate steps in its own Workflow (`:32` ≡ step 9, `:33` ≡ step 9, `:36` ≡ step 2,
|
||||
`:31` ≡ the description). Its body is 1,102 words and its whole file 1,217, so it does fail the
|
||||
900-word body FAIL — but for its length, not for the restatement. The four duplicated Gotchas are 114
|
||||
words between them; delete every one and the file still fails, while a skill 250 words shorter with
|
||||
the identical defect passes clean. The two properties are uncorrelated, which is why the counts are a
|
||||
backstop to the dispatch rule and the Gotchas constraint — both of which are auditor judgment for the
|
||||
semantic half, per the Enforcement table — and not a substitute for them. Reading the word gate as
|
||||
the mechanism is the specific mistake this paragraph exists to prevent.
|
||||
|
||||
**Some skills legitimately need more description budget than others.** A tiered limit keyed to
|
||||
sibling density was considered and rejected as too clever; the flat 250/400 pair means the `gitea-*`
|
||||
and `git-*` families — where every sibling shares a keyword and boundary clauses do real routing work
|
||||
— are the ones most likely to sit at the FAIL tier permanently. If the retrofit shows that family
|
||||
routing degrades, the tier is the first thing to revisit.
|
||||
|
||||
**Four broken routing targets were found; two are fixed here and two are live.** Tracked as issue
|
||||
#100.
|
||||
|
||||
- `skill-audit` routed to `/skill-improve` twice in its description plus `README.md:10`, and no such
|
||||
skill exists — the real target is `skill-author`. **Fixed here**, as a side effect of retrofitting
|
||||
kyberforge's own skills.
|
||||
- `agent-author` said "Do not use for read-only review — examine agent files manually", routing away
|
||||
from `agent-audit`, the correct sibling. **Fixed here**, same way. Note this one was never
|
||||
detectable by the resolvable-target check and never will be: "examine agent files manually" names
|
||||
no target, and a check that resolves names cannot see a name that is absent. A misroute to nowhere
|
||||
is a review finding, not a gate finding.
|
||||
- `research` routes to `neuledge-context`, which exists only inside that string. **Live.**
|
||||
- `gitea-issues` carries the literal string `gitea-labels- milestones` in its folded description, a
|
||||
stray space introduced by YAML wrapping mid-token, breaking the skill name in preloaded text.
|
||||
**Live** — the check reports it as a dangling `gitea-labels`.
|
||||
|
||||
So the check fires on 3 of the 4 against the base commit and on 2 at the tip of this change, and
|
||||
`tests/test-skill-size-check.sh` probes exactly those three by name rather than asserting a count, so
|
||||
it degrades to SKIP as #100 lands rather than going stale.
|
||||
|
||||
**Duplication between `skill-author` and `agent-author` survives un-gated.** The merge rule
|
||||
deliberately excludes the author pair, so the commit-verification argument in four near-copies, the
|
||||
root-cause grouping rule in four copies, and the wholesale clone of the "Improving an existing X"
|
||||
flow all remain. Cache isolation makes them structurally unavoidable
|
||||
(`skill-audit/SKILL.md:95` forbids cross-skill references; `LESSONS.md:107` records why), so the
|
||||
options are a sync gate or continued drift. This is an input to issue #101, which carries both halves
|
||||
of the kyberforge duplication problem — the deferred audit-pair merge and this — not a solved
|
||||
problem.
|
||||
|
||||
**Provenance frontmatter is explicitly out of scope.** `LESSONS.md:63` asserts that non-routing
|
||||
frontmatter (`source_keys`, `category`, `version`) is loaded at agent startup, which would make the
|
||||
7,352 characters of it across the corpus a third again on top of the description tax. Measured
|
||||
against a live session on this Claude Code version, it is not: the model-visible skill listing
|
||||
contains only `name` and `description`. That is host-observed rather than spec-guaranteed and says
|
||||
nothing about Copilot CLI, but it is sufficient to establish that cutting `source_keys` would break
|
||||
the ADR-0009 provenance machinery for no runtime gain. The metadata was added deliberately and stays.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
Upstream citations below are relative to
|
||||
`plugins/kyberforge/docs/research/examples/skill-write/`, as in Context above.
|
||||
|
||||
- **Keep pushiness, raise the budget to ~500 chars.** Undertriggering is the worse failure mode — a
|
||||
skill that never fires is worth nothing regardless of cost — and `skill-creator/SKILL.md:67`
|
||||
explicitly recommends being "pushy" against an observed undertriggering tendency. Rejected because
|
||||
that claim is an unmeasured assertion about an older model, and because the correctness hazard in
|
||||
`writing-skills/SKILL.md:154-158` cuts the other way: a fat description is not merely expensive, it is a
|
||||
shortcut agents take instead of reading the body. Would have landed a 35% cut.
|
||||
- **A trigger-eval loop to set lengths empirically.** `skill-creator/SKILL.md:337-404` specifies 20
|
||||
queries per skill, 8-10 positive and 8-10 near-miss, with a 60/40 train/test split selecting on
|
||||
test score. This is the rigorous answer and the repo has deliberately never built it. Rejected
|
||||
because it blocks the context cut behind a substantial new subsystem.
|
||||
- **A repo-level aggregate preload budget** (≤12,000 chars across all skills, checked at pre-push).
|
||||
The only option that measures the actual goal rather than a proxy. Rejected because it makes one
|
||||
skill's edit fail on account of another skill's growth, and because it is meaningless for an
|
||||
external consumer installing a subset of the plugins.
|
||||
- **500-word body FAIL, matching `writing-skills/SKILL.md:217-221`.** Best-grounded in upstream and
|
||||
would align this repo with the tightest source. Rejected because it fails 28 of 39 skills body-only
|
||||
(35 of 39 measured whole-file), and a blunt gate gets satisfied by deleting content rather than
|
||||
relocating it.
|
||||
- **A shrinking baseline file** recording each non-compliant skill's current numbers, failing only on
|
||||
growth. Would have made the retrofit a visible burn-down instead of a wall. Rejected in favour of
|
||||
hot gates.
|
||||
- **A sync gate over the duplicated spans** instead of a merge rule — generalising
|
||||
`scripts/check-vale-style-sync.sh` to cover shared prose so duplication persists but drift cannot.
|
||||
Rejected for the audit pair in favour of merging, which removes the duplication rather than
|
||||
policing it, and removes a mutually-excluding near-miss pair from the router at the same time. It
|
||||
remains the only available answer for the author pair.
|
||||
- **Merging `skill-author` + `agent-author` as well**, taking kyberforge from seven skills to five.
|
||||
Largest cut available. Rejected because it reopens ADR-0005, ADR-0008 and ADR-0016 together, and a
|
||||
merged author skill would carry both the skill-directory scaffold and the dual-provider agent
|
||||
scaffold behind one dispatch.
|
||||
- **Demoting Gotchas** to the end of the body or into `references/gotchas.md`, removing its
|
||||
position-based exemption from the dispatch rule. Maximum saving on the largest body construct
|
||||
(6,830 words, 21% of all body text). Rejected because a gotcha read after the mistake is worthless.
|
||||
227
docs/adr/0021-plugin-descriptions-state-a-domain-boundary.md
Normal file
227
docs/adr/0021-plugin-descriptions-state-a-domain-boundary.md
Normal file
@@ -0,0 +1,227 @@
|
||||
# A plugin's published description states its domain boundary and never enumerates its skills
|
||||
|
||||
Three of this repo's six plugins publish a `description` that lists the skills they ship. That style
|
||||
has now failed three times in four days, the third time inside the correction for the second. It is
|
||||
enforced by nothing, it obliges a marketplace release on every skill addition, and it was never
|
||||
applied to the other three plugins. This ADR retires it: a published description says what the
|
||||
plugin is *for*, and the inventory lives where an inventory can be read off the tree.
|
||||
|
||||
**Status: accepted (2026-08-17).**
|
||||
|
||||
## Context
|
||||
|
||||
A plugin's published description is one string authored twice — in `plugins/<name>/apm.yml` and in
|
||||
the matching `marketplace.packages[]` entry of the root `apm.yml` — and compiled into four generated
|
||||
files per plugin edit: the plugin's `.claude-plugin/plugin.json` and `.github/plugin/plugin.json`,
|
||||
plus the repo-wide `.claude-plugin/marketplace.json` and its `.github/plugin/marketplace.json`
|
||||
mirror. (`.agents/plugins/marketplace.json`, apm's codex profile, carries no per-package
|
||||
`description` or `version` at all and is unaffected.) It is the only text a consumer sees in a marketplace listing before
|
||||
installing. It is **not** a SKILL.md `description`: it is never preloaded into an agent's context and
|
||||
routes nothing at runtime. ADR-0020 governs that other artifact; this one governs this one. The
|
||||
overlap is a finding, not a scope: ADR-0020 established that capability enumeration in a description
|
||||
is "a correctness hazard, not only a token cost". The hazard at this layer is different — staleness
|
||||
in published metadata rather than an agent shortcutting the body — but the enumeration is the same
|
||||
construct and it fails the same way.
|
||||
|
||||
Measured at `de84d1b`, the branch tip before this change. Each figure is reproducible from the tree:
|
||||
skill counts are `ls plugins/<name>/.apm/skills/ | wc -l`, description text is
|
||||
`plugins/<name>/apm.yml`.
|
||||
|
||||
| Plugin | Style | Skills | Items enumerated | Skills named | Unnamed |
|
||||
|---|---|---|---|---|---|
|
||||
| `bin` | enumeration | 11 | 8 | 9 | `caveman`, `zoom-out` |
|
||||
| `git` | enumeration | 9 | 8 | 8 | `git-workflow` |
|
||||
| `gitea` | enumeration | 7 | 7 | 6 | `gitea-workflow` |
|
||||
| `core` | boundary | 3 | — | — | — |
|
||||
| `kyberforge` | boundary | 7 | — | — | — |
|
||||
| `lint` | boundary | 2 | — | — | — |
|
||||
|
||||
Three failures, in order.
|
||||
|
||||
**`bb9158d` (2026-08-14) — `core`'s description described `bin`.** The text it deleted read
|
||||
"Cross-cutting utility skills for everyday AI-assisted coding — triage, diagnosis, architecture
|
||||
review, and session navigation." All four items are real skills and not one of them is `core`'s:
|
||||
they are `bin`'s `triage`, `diagnose`, `improve-codebase-architecture` and `zoom-out`. `core` ships
|
||||
`agentsmd-author`, `agentsmd-audit` and `provider-adapter-author`, and the published description
|
||||
named none of them.
|
||||
|
||||
This is the failure the whole style was later adopted against, and it is worth being exact about
|
||||
what it was, because the record has been read the other way twice since. It was **wrong content**,
|
||||
not an incomplete list. The description was a syntactically perfect, complete, four-item enumeration
|
||||
of a real skill set; it just belonged to a different plugin. Enumerating harder could not have caught
|
||||
it, and a gate that asked "does every enumerated item exist as a skill?" would have passed it — all
|
||||
four did exist. `bb9158d`'s own fix went the other direction: it replaced the enumeration with a
|
||||
domain boundary, and `core` has needed no correction since. The precedent set by that commit was
|
||||
therefore *boundary*, and the two commits below cite it while doing the opposite.
|
||||
|
||||
**`65bac15` (2026-08-17) — `git` advertised `gitea`'s domain, `gitea` advertised a skill that does
|
||||
not exist.** `git` read "conventional commits, branch management, pull requests, and feature flow";
|
||||
pull requests reach the forge over HTTP and are `gitea`'s, which is the exact boundary
|
||||
`docs/spec/architecture.md` draws between the two plugins. `gitea` read "issues, pull requests,
|
||||
milestones, releases, and wikis"; `grep -ri wiki plugins/gitea/.apm/` returns nothing and no wiki
|
||||
skill has ever existed. Both were repaired by re-enumerating.
|
||||
|
||||
**`de84d1b` (2026-08-17) — the re-enumeration was itself incomplete.** `bin`'s "A place for things to
|
||||
be binned" was replaced with an eight-item list over eleven skills; `caveman` and `zoom-out` are
|
||||
absent. `zoom-out` is the same skill `bb9158d` had called "session navigation" three days earlier
|
||||
while deleting it from the wrong plugin's description — named when it was in the wrong place,
|
||||
unnamed once it was in the right one. And the miss is not confined to `bin`: `git-workflow` is
|
||||
unnamed in `git`'s corrected description, though `65bac15`'s own commit message states it was added
|
||||
("omitting pc-author/pc-run, git-submodules and git-workflow"), and `gitea-workflow` is unnamed in
|
||||
`gitea`'s. Across the three plugins, 23 of 27 skills are named at the third attempt.
|
||||
|
||||
**Nothing checks any of this.** `scripts/check-manifests.sh` does not contain the string
|
||||
`description`. The three ADR-0020 validators (`scripts/skill-size-check.sh` and skill-audit's and
|
||||
agent-audit's `validate.sh`) gate on SKILL.md and agent frontmatter; they do open `apm.yml`, but only
|
||||
to read `dependencies.apm` when resolving the boundary-target universe — none of them reads the
|
||||
`description:` key, and their hook globs match `SKILL.md` and `*.agent.md` only. `apm audit --ci`,
|
||||
`apm pack --check-clean` and `scripts/sync-plugin-content.sh --check --all` all compare compiled
|
||||
output against `apm.yml`, so their entire job is to propagate whatever the description says into
|
||||
those four files byte-for-byte and confirm they match. The `wiki` claim passed every one of the fourteen pre-push hooks, every day it
|
||||
was published.
|
||||
|
||||
**And the obligation is unbounded.** Under enumeration, adding one skill to `bin`, `git` or `gitea`
|
||||
means editing two copies of a prose string on top of the version bumps and regeneration any skill
|
||||
addition already owes under this repo's release policy
|
||||
(`plugins/kyberforge/.apm/skills/apm-workflow/references/marketplace.md`). The bumps are not the
|
||||
marginal cost — the prose edit is, and it is the half nothing checks. A skill *rename* triggers the
|
||||
same, for a string no consumer can tell went stale. 27 of the repo's 39 skills sat behind
|
||||
a description carrying that obligation; the other 12 did not, and their three plugins have generated
|
||||
no defect of this class.
|
||||
|
||||
### Scope
|
||||
|
||||
This decision covers the six plugins this repo authors. The root marketplace also lists
|
||||
`mattpocock-skills`, a third-party package whose description is not this repo's to write; its entry
|
||||
is out of scope and is left as published upstream.
|
||||
|
||||
## Decision
|
||||
|
||||
**A plugin's published `description` states the plugin's domain boundary. It does not enumerate the
|
||||
skills the plugin ships, by name or by paraphrase.**
|
||||
|
||||
- The boundary answers "what kind of work belongs to this plugin, and where is its edge against its
|
||||
nearest sibling" — the question a consumer deciding whether to install is actually asking. It is
|
||||
stable under skill addition, rename and removal, which is the entire point: an artifact that does
|
||||
not change when the tree changes cannot go stale against it.
|
||||
- **The boundary must cover everything the plugin actually ships.** A boundary drawn narrower than
|
||||
the contents is the same defect as an incomplete enumeration, one level up, and it is the specific
|
||||
risk in this change. `git` carries `pc-author` and `pc-run`, which are not git operations at all;
|
||||
"Skills for working with Git" silently drops them, so the boundary names the pre-commit hooks
|
||||
explicitly rather than trusting a reader to file them under Git.
|
||||
- The two copies — package `apm.yml` and the root `marketplace.packages[]` entry — stay identical.
|
||||
This is already the rule in practice and both prior corrections state why: the root entry is what
|
||||
reaches the compiled marketplace, so fixing only the package manifest leaves it half-propagated.
|
||||
- The three descriptions, rewritten here, with `core`/`kyberforge`/`lint` shown for register:
|
||||
|
||||
| Plugin | Published description | Chars |
|
||||
|---|---|---|
|
||||
| `bin` | Skills for everyday AI-assisted development work that is not tied to a single tool, forge or language, and has not yet been split into a focused plugin. | 152 |
|
||||
| `git` | Skills and agents for working with a local Git clone over the git wire protocol, and for authoring and running the pre-commit hooks that guard it. | 146 |
|
||||
| `gitea` | Skills and agents for working with a Gitea forge through its HTTP API — the forge's own objects, as distinct from the local git clone. | 134 |
|
||||
| `core` | *(unchanged)* Skills for authoring and auditing a repo's AGENTS.md and the provider adapter files that defer to it. | 101 |
|
||||
| `kyberforge` | *(unchanged)* Skills and agents for creating, maintaining, and managing a Claude Code / Copilot CLI plugin marketplace. | 105 |
|
||||
| `lint` | *(unchanged)* Skills and agents for configuring and running linters. | 54 |
|
||||
|
||||
- **No gate is added.** This is a deliberate omission and the reasoning is below, not an item left
|
||||
for later.
|
||||
|
||||
### Why no gate
|
||||
|
||||
The check enumeration would need — "every skill directory appears in the description" — was writable
|
||||
in principle and was never written, including by the two commits that corrected an enumeration by
|
||||
enumerating again and had every reason to. It is also only half a check: it
|
||||
catches a skill missing from the list, and it cannot catch `wiki`, because "this noun does not name
|
||||
any skill" requires a vocabulary of permissible non-skill nouns that no one is going to maintain.
|
||||
Under a boundary there is no correspondence left to check, which is the property being bought.
|
||||
|
||||
What survives un-gated is `bb9158d`'s actual failure: a boundary that is simply wrong about its
|
||||
plugin. That was never machine-checkable in either style — the text was a well-formed description of
|
||||
a real plugin — and it is caught by the same review that has to happen when a published,
|
||||
consumer-facing string is edited at all. A gate that would catch it needs a declared per-plugin
|
||||
skill-to-boundary mapping for the description to be checked against, which is a second artifact
|
||||
requiring exactly the per-skill maintenance this ADR exists to delete, relocated one file over.
|
||||
|
||||
Two cheap partial gates were considered and rejected in the same breath. Forbidding a comma-separated
|
||||
run of three or more noun phrases is a prose heuristic that fires on `lint`'s perfectly good
|
||||
"configuring and running linters" class of sentence. Forbidding any string matching a skill directory
|
||||
name under `plugins/<name>/.apm/skills/` bans legitimate boundary vocabulary — `git-branches` exists,
|
||||
and a `git` boundary has every right to say "branches". Both would be believed, and both would be
|
||||
wrong, which ADR-0020 already records as worse than no gate.
|
||||
|
||||
## Considered options
|
||||
|
||||
**Keep enumeration and gate it.** The only option that makes the current style safe. Rejected on the
|
||||
three grounds above: the check is one-directional, it cannot see an invented capability, and it makes
|
||||
a marketplace release the consequence of adding a directory. It also hard-couples published consumer
|
||||
copy to internal directory names, so a skill rename becomes a version bump on the plugin and on the
|
||||
marketplace.
|
||||
|
||||
**Enumerate consistently across all six plugins**, on the grounds that the real defect is the split
|
||||
style. Rejected: it takes an obligation that has produced three failures on three plugins and applies
|
||||
it to six. The measured outcome of the most recent attempt to enumerate carefully, with the defect
|
||||
fresh and two prior commits as precedent, is four skills unnamed.
|
||||
|
||||
**Cap the description length**, mirroring ADR-0020's 250/400-character tiers, on the theory that a
|
||||
short description has no room to enumerate. Rejected because length does not measure correspondence:
|
||||
`gitea`'s failing description was 96 characters and asserted a skill that has never existed, while
|
||||
`bin`'s 176-character enumeration is under the same cap. All six descriptions here, before and after,
|
||||
sit inside ADR-0020's tiers; the tier would have been silent through all three failures.
|
||||
|
||||
**Delete the description to a bare name.** Rejected: apm's Claude marketplace mapper emits
|
||||
`description` into `marketplace.json`, and it is the only prose a consumer sees before installing.
|
||||
|
||||
**Point the description at the plugin's `README.md`.** Rejected: a marketplace listing renders a
|
||||
string, not a link — and the README's own plugin list carries the same enumeration with the same
|
||||
staleness, so this relocates the defect rather than fixing it.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Three descriptions are rewritten and the compiled output regenerated.** Eight generated files
|
||||
change: `plugins/{bin,git,gitea}/.claude-plugin/plugin.json`,
|
||||
`plugins/{bin,git,gitea}/.github/plugin/plugin.json`, `.claude-plugin/marketplace.json` and its
|
||||
byte-identical `.github/plugin/marketplace.json` mirror. `.agents/plugins/marketplace.json` (the
|
||||
codex profile) is unchanged and correctly so — it carries no per-package `description` or `version`
|
||||
field at all, only `name`, `source`, `policy` and `category`.
|
||||
|
||||
**Version bumps, all PATCH under the `per_package` strategy:** `bin` 1.1.4 → 1.1.5, `git` 1.3.4 →
|
||||
1.3.5, `gitea` 1.3.5 → 1.3.6, `marketplace.version` 0.4.4 → 0.4.5.
|
||||
|
||||
**The root `apm.yml` top-level `version:` is restored to lockstep with `marketplace.version`,
|
||||
0.4.2 → 0.4.5.** These two fields have moved together in every commit that has ever touched root
|
||||
`apm.yml` — 0.3.2, 0.3.3, 0.3.4, 0.4.0, 0.4.1, 0.4.2 in both — until `65bac15` and
|
||||
`de84d1b` on this branch bumped `marketplace.version` to 0.4.3 and then 0.4.4 while leaving the
|
||||
top-level field at 0.4.2. Lockstep is not folklore: it is stated at
|
||||
`plugins/kyberforge/.apm/skills/apm-workflow/references/marketplace.md`. This is a defect, not a
|
||||
style: `apm.yml`'s comment inside the marketplace block records that the top-level `version:` is not inherited into the compiled output
|
||||
"despite being used elsewhere (e.g. by `apm audit`)", so the field is live and was silently two
|
||||
releases behind what the marketplace published. Closed here rather than tracked, because the
|
||||
correction is one line and the drift is three days old.
|
||||
|
||||
**`docs/spec/architecture.md`'s plugin table is unchanged and stays a routing table.** It answers
|
||||
"where does a new skill go" for someone working *inside* this repo; the published description answers
|
||||
"should I install this" for someone outside it. The two now read similarly, and that is not
|
||||
duplication to collapse — they have different readers and different lifecycles, and the table already
|
||||
says so in its own preamble ("These are routing boundaries, not inventories"). One caveat for whoever
|
||||
next edits that page: its closing sentence sends a reader to the published description "for what a
|
||||
consumer actually gets", which was true against an enumeration and is now a pointer to a second
|
||||
boundary statement. Neither artifact carries an inventory after this change, so that sentence was
|
||||
rewritten in the same branch to point at `plugins/<name>/.apm/skills/` and `README.md` instead.
|
||||
|
||||
**`README.md`'s plugin bullet list becomes the only place an inventory lives, and it still
|
||||
enumerates.** That is deliberate, but it makes the list load-bearing in a way it was not before, so
|
||||
its `bin`, `git` and `gitea` bullets were completed in the same branch to name every skill those
|
||||
plugins ship. This ADR does not otherwise extend to it: a README is a hand-read document where a
|
||||
list of what you get is the useful thing, it is not compiled into four files, and a stale line in it
|
||||
costs a reader a moment rather than misrepresenting a published package. The tradeoff that makes
|
||||
enumeration wrong in a marketplace manifest is precisely the one that makes it fine there.
|
||||
|
||||
**Nothing in the ADR-0020 gate set changes.** Its character and word tiers, its Vale rules and its
|
||||
three validators all read `SKILL.md` and `*.agent.md` frontmatter; none of them opens an `apm.yml`.
|
||||
The two contracts are adjacent and independent, and a future author retrofitting a skill under
|
||||
issue #99 is not touched by this ADR.
|
||||
|
||||
**The failure mode this leaves open is a wrong boundary, and it is un-gated by design.** If a fourth
|
||||
failure of this class occurs it will be a description that describes the wrong plugin — `bb9158d`'s
|
||||
shape, the one enumeration never addressed. That is the trigger to revisit, and the thing to build
|
||||
then is a declared skill-to-boundary mapping, not a return to enumeration.
|
||||
@@ -19,25 +19,44 @@ project repo (local overrides)
|
||||
- **Executables** (`DEPLOY_EXECUTABLES`): `providers/claude-code/statusline-command.sh` → `~/.claude/statusline-command.sh` (with `+x`)
|
||||
- **Directories** (`DEPLOY_DIRS`): `core/` → `~/.claude/core/` (destination fully replaced on each deploy)
|
||||
|
||||
Skills are **not** deployed by `install.sh`. They are distributed as plugins and installed separately via `claude plugin install <name>@holocron`.
|
||||
Skills are **not** deployed by `install.sh`. They are distributed as plugins and installed separately — in this repo by `apm install` against the `dependencies.apm` entries in the root `apm.yml`, which lands them in `.claude/skills/` and `.claude/agents/` (ADR-0018); elsewhere by `claude plugin install <name>@holocron`.
|
||||
|
||||
`~/.claude/CLAUDE.md` is a thin adapter, not a content source. It imports `~/.agents/AGENTS.md` (always-on rules) and `governance.md` (always-on governance), then lists the content index. All always-on content lives in `AGENTS.md` files so other providers can import the same source without duplication.
|
||||
`~/.claude/CLAUDE.md` is a thin adapter, not a content source. It imports `~/.agents/AGENTS.md` (always-on rules) and `governance.md` (always-on governance) and carries nothing else — the content index of on-demand instruction files sits in `core/AGENTS.md`, deployed to `~/.agents/AGENTS.md` and imported by it. All always-on content lives in `AGENTS.md` files so other providers can import the same source without duplication.
|
||||
|
||||
## Plugin model
|
||||
|
||||
Skills, agents, MCP servers, and hooks are distributed as self-contained plugin units under `plugins/`, installed independently via `claude plugin install <name>@holocron`. Each plugin is an **apm package**: `plugins/<name>/apm.yml` plus a hand-authored `plugins/<name>/.apm/{skills,agents,hooks,commands,instructions,extensions}/` tree (ADR-0015). There is no hand-maintained `plugin.json` — every manifest and every host-visible content directory is compiled from that source.
|
||||
Skills, agents, MCP servers, and hooks are distributed as self-contained plugin units under `plugins/`, installed independently — via `apm install` here, or `claude plugin install <name>@holocron` for a host consuming the marketplace natively (ADR-0018). Self-contained is a hard constraint, not a description: a plugin is copied to a cache on install, so nothing inside it may reference a file outside its own directory. That is why the Vale styles are duplicated across two skills rather than shared (ADR-0014), and why ADR-0020's constants are copied into three validators rather than sourced from one. Each plugin is an **apm package**: `plugins/<name>/apm.yml` plus a hand-authored `plugins/<name>/.apm/{skills,agents,hooks,commands,instructions,extensions}/` tree (ADR-0015). There is no hand-maintained `plugin.json` — every manifest and every host-visible content directory is compiled from that source.
|
||||
|
||||
Which plugin a new skill belongs in follows from what each one is scoped to. The boundary that matters most in practice is `core` vs `kyberforge`: `core` is the home for cross-cutting, repo-agnostic utility skills that a consumer would want against *their* repo, while `kyberforge` is meta-tooling for the holocron marketplace itself. A skill that authors a target repo's `AGENTS.md` is `core`; a skill that audits a `SKILL.md` against this marketplace's contract is `kyberforge`.
|
||||
|
||||
The second boundary worth stating is `git` vs `gitea`, because both own things called branches and both touch pull requests: `git` is whatever works over the git wire protocol against a local clone, `gitea` is whatever goes through the forge's HTTP API. That is why `git-branches` and `gitea-branches` both exist and are not duplicates.
|
||||
|
||||
These are routing boundaries, not inventories — they answer "where does a new skill go", so they deliberately do not enumerate what each plugin ships today. The plugin's published `description` in its `apm.yml` states the same boundary for a consumer deciding whether to install (ADR-0021); neither carries an inventory. For what a plugin ships today, read `plugins/<name>/.apm/skills/` or the plugin list in `README.md`.
|
||||
|
||||
| Plugin | Scope |
|
||||
|---|---|
|
||||
| `core` | Authoring and auditing a repo's `AGENTS.md` and the provider adapter files that defer to it |
|
||||
| `git` | Git operations and git hook tooling — anything driven over the git wire protocol against a local clone, plus the pre-commit hooks that guard it |
|
||||
| `gitea` | Anything reached through the Gitea HTTP API rather than the git wire protocol — the forge's own objects |
|
||||
| `kyberforge` | Creating and maintaining a Claude Code / Copilot CLI plugin marketplace — this repo's own meta-tooling |
|
||||
| `lint` | Configuring and running linters against a target repo; repo-agnostic, first linter is Vale |
|
||||
| `bin` | Unsorted skills that have not earned a home yet |
|
||||
|
||||
Two compilers produce the plugin roots you see in the tree:
|
||||
|
||||
- **`apm pack` compiles the manifests** (ADR-0015). Per plugin: `.claude-plugin/plugin.json` and `.github/plugin/plugin.json`, both generated from `plugins/<name>/apm.yml`. Repo-wide, from the root `apm.yml`'s `marketplace:` block: `.claude-plugin/marketplace.json` (apm's `claude` output profile) and `.agents/plugins/marketplace.json` (its `codex` profile, a differently-shaped file). Those two are the only marketplace outputs apm has profiles for — the third root manifest, `.github/plugin/marketplace.json` (Copilot CLI's legacy path), is a byte-identical mirror of the Claude one maintained by `scripts/sync-marketplace-mirror.sh` and gated by the `check-marketplace-mirror-sync` pre-push hook.
|
||||
- **`scripts/sync-plugin-content.sh` compiles the content mirror** (ADR-0017). It wraps `apm pack --format plugin` and copies the resulting bundle's flat `agents/`, `skills/`, `commands/`, `instructions/`, `extensions/`, and merged `hooks/hooks.json` back to the plugin root. Claude Code's installer convention-scans those flat paths and has no `.apm/` awareness whatsoever, so the mirror exists solely to satisfy the host's discovery contract.
|
||||
|
||||
`.apm/` is the sole hand-edited authoring source for plugin content. An edit made in the flat mirror is discarded by the next sync and is reported as drift by the `check-plugin-content-sync` pre-push hook. Hand-authored material that is not an `.apm/` primitive — `README.md`, `docs/`, `bin/`, `sources.md`, and `.mcp.json` — lives at the plugin root and is untouched by either compiler.
|
||||
`.apm/` is the sole hand-edited authoring source for plugin content. An edit made in the flat mirror is discarded by the next sync and is reported as drift by the `check-plugin-content-sync` pre-push hook. Hand-authored material that is not an `.apm/` primitive — `README.md`, `docs/`, `bin/`, `sources.md`, `.mcp.json`, and per-plugin extras such as `plugins/git/config.example.json`, `plugins/gitea/references/` and `plugins/bin/evals/` — lives at the plugin **root** and is untouched by either compiler.
|
||||
|
||||
That immunity is positional, not by filename. Anything placed *inside* a mirrored directory is destroyed regardless of what it is: `sync_dir` runs `rm -rf "$dst"` before every copy, and `sync_hooks_json` does the same to `hooks/`. A hand-written `README.md` under `plugins/<name>/hooks/` or `plugins/<name>/skills/` is deleted by the next sync with no drift report, because a file with no `.apm/` counterpart is simply absent from the regenerated tree. This has already cost the repo one document — `plugins/kyberforge/hooks/README.md`, since restored to `plugins/kyberforge/docs/hooks.md`. Plugin-root documentation belongs in `docs/`.
|
||||
|
||||
## Governance layer
|
||||
|
||||
`core/instructions/governance.md` is the always-on governance instruction file. Unlike the on-demand instruction files in the content index, governance.md is loaded into every Claude session via `@import` in `providers/claude-code/CLAUDE.md`. This is a technical guarantee, not a behavioural instruction — `@import` causes Claude Code to expand and load the file at launch, before any interaction begins.
|
||||
|
||||
Those on-demand files are plain markdown — no frontmatter, no schema. The agent decides when to read each one from task context and the content index label alone. Frontmatter is deferred until there is evidence that agents are loading the wrong files in practice; it is a deliberate deferral, not an oversight to close.
|
||||
|
||||
The governance layer has two phases:
|
||||
- **Phase 1** (complete): instruction and documentation layer — `governance.md` loaded via `@import`; `docs/ai-constitution.md` and `docs/wiki/HUMANS.md` as human-facing reference; `CONTEXT.md` extended with governance domain language.
|
||||
- **Phase 2** (planned): deterministic enforcement layer — pre-commit hooks, CI gates, secret scanning, licence scanning. Specified in `docs/research/governance_principles/CONTROLS.md`.
|
||||
@@ -51,7 +70,13 @@ This repo uses two `AGENTS.md` files as the provider-agnostic source of always-o
|
||||
|
||||
Both `CLAUDE.md` files are thin adapters: they import from their respective `AGENTS.md` and add only Claude Code-specific syntax (`@import`, content index paths). They carry no original always-on content.
|
||||
|
||||
This repo also has a `CLAUDE.md` at its root — the Claude Code entry point for working in this repo. It imports `AGENTS.md` and `CONTEXT.md`, nothing more. This is distinct from `providers/claude-code/CLAUDE.md`, which is the global config deployed to `~/.claude/`.
|
||||
This repo also has a `CLAUDE.md` at its root — the Claude Code entry point for working in this repo. It imports `AGENTS.md` and nothing else; there is no `@CONTEXT.md` import. It is not import-only either: below the import sits a fenced `<!-- rtk-instructions v2 -->` … `<!-- /rtk-instructions -->` block carrying the RTK command-prefix convention, which is tool-specific content with no `AGENTS.md` source. This is distinct from `providers/claude-code/CLAUDE.md`, which is the global config deployed to `~/.claude/`.
|
||||
|
||||
`CONTEXT.md` is therefore **not** always-loaded. `AGENTS.md` instructs agents to read it at session start, which is a behavioural instruction, not an `@import` guarantee — `LESSONS.md`'s 2026-05-17 entry proposed adding the import and it was never applied. Treat that entry as open work rather than a record of a landed change.
|
||||
|
||||
## Reference conventions
|
||||
|
||||
The stated convention is that files referencing other files declare those references explicitly: the referencing file carries the forward reference (the content index in `core/AGENTS.md`, `references:` in frontmatter), the referenced file carries a `when:` field describing when it is loaded, and divergence between the two signals staleness. It is aspirational, not a description of the repo today — no file under `core/instructions/` carries frontmatter at all, `when:` appears in exactly one of the 39 `SKILL.md` sources under `plugins/*/.apm/skills/`, and the reference scanner script meant to derive the reverse map ("what files reference this file?") does not exist; `docs/notes/skill-implementation-workflow.md` still lists it as unbuilt work. Treat it as intent for instruction files, skills, and workflow documents, not as a rule the repo enforces.
|
||||
|
||||
## Provider model
|
||||
|
||||
@@ -59,4 +84,4 @@ This repo also has a `CLAUDE.md` at its root — the Claude Code entry point for
|
||||
|
||||
## Architectural decisions
|
||||
|
||||
Key hard-to-reverse decisions are recorded as ADRs in `docs/adr/`. See the index there for rationale on choices like the pull distribution model, copy-not-symlink coupling, and the two-tier CLAUDE.md structure.
|
||||
Key hard-to-reverse decisions are recorded as ADRs in `docs/adr/`. There is no index file — the directory holds numbered ADRs whose filenames state their decision, so `ls docs/adr/` is the index. Read a superseding ADR before the one it supersedes: ADR-0015 (apm as the authoring source of truth) supersedes ADR-0001 and moots ADR-0006, ADR-0017 corrects ADR-0015's host-discovery gap, and ADR-0019 supersedes one claim in ADR-0018 (that `.claude/settings.json`'s committed content is exactly `{"hooks": {}}`) while keeping the rule behind it. Entry points for the structure described on this page: ADR-0002 (two-tier CLAUDE.md), ADR-0003 (AGENTS.md as the provider-agnostic entry point), ADR-0015 and ADR-0017 (the two compilers behind the plugin roots).
|
||||
|
||||
689
docs/spec/gates.md
Normal file
689
docs/spec/gates.md
Normal file
@@ -0,0 +1,689 @@
|
||||
# Enforcement gates
|
||||
|
||||
Reference for this repo's pre-commit and pre-push hooks: what each one guards, what its numbers
|
||||
mean, and which shapes were tried and rejected. Read it when a gate fails, before changing anything
|
||||
in `.pre-commit-config.yaml`, or before "fixing" something that looks like an inconsistency — several
|
||||
of the oddities documented here are load-bearing and have already been re-litigated once.
|
||||
|
||||
`AGENTS.md` carries only the operative rules an agent needs in the moment. The reasoning lives here.
|
||||
|
||||
---
|
||||
|
||||
## Running the gates
|
||||
|
||||
| Command | Scope |
|
||||
|---|---|
|
||||
| `pre-commit run --all-files` | the commit-stage hooks |
|
||||
| `pre-commit run --hook-stage pre-push --all-files` | the push gate, one command — with one caveat below |
|
||||
| `pre-commit run skill-size-check --all-files` | just the ADR-0020 size/context gates |
|
||||
|
||||
Install hooks via `pc-run`, wiring **all three stages**. This repo's `.pre-commit-config.yaml` has no
|
||||
`default_install_hook_types`, so a plain install silently skips `commit-msg` (Conventional Commits)
|
||||
and `pre-push` (everything below).
|
||||
|
||||
The pre-push command reports **16** hooks, not 14. The extra two are pre-commit's own `meta` hooks,
|
||||
`check-hooks-apply` and `check-useless-excludes`: they declare no `stages:`, so they run at every
|
||||
stage including this one. Both are declared in this repo's `.pre-commit-config.yaml` like everything
|
||||
else — what separates them is `repo: meta` (pre-commit's own built-ins) from `repo: local`. Fourteen
|
||||
is the count of hooks this repo authors itself.
|
||||
|
||||
**The caveat: one of those 14 is a silent no-op under that invocation.**
|
||||
`check-release-needed` exits 0 immediately unless `PRE_COMMIT_REMOTE_BRANCH` equals
|
||||
`refs/heads/main`, and pre-commit exports that variable only from the real pre-push git hook during
|
||||
an actual `git push`. Running the stage by hand — or from a CI runner — therefore reports it
|
||||
`Passed` having checked nothing. That is by design for feature branches — pushing WIP must not be
|
||||
blocked on cutting a premature tag — but it means `--hook-stage pre-push --all-files` is a full
|
||||
rehearsal of 13 hooks and a skip of the fourteenth. The script's own header records the same gap for
|
||||
a PR merged through Gitea's merge button, where no local push happens at all.
|
||||
|
||||
## The pre-push gate
|
||||
|
||||
Fourteen hooks, grouped below by what they guard rather than by the order `.pre-commit-config.yaml` declares them in.
|
||||
|
||||
**Core checks**
|
||||
|
||||
| Hook | Guards |
|
||||
|---|---|
|
||||
| `run-tests` | `bash tests/run-tests.sh --strict` — the whole suite, skips fatal (see [Tests](#tests)) |
|
||||
| `check-manifests` | `marketplace.json` and `plugin.json` paths resolve (needs `jq`) |
|
||||
|
||||
**Generated-content drift gates**
|
||||
|
||||
| Hook | Guards |
|
||||
|---|---|
|
||||
| `check-plugin-content-sync` | each plugin's flat `skills/agents/commands/hooks` mirror matches `.apm/` (issue #90) |
|
||||
| `check-marketplace-mirror-sync` | `.github/plugin/marketplace.json` is byte-identical to `.claude-plugin/marketplace.json` — no apm output profile targets that path |
|
||||
| `check-vale-style-sync` | skill-audit's Vale copy matches agent-audit's canonical copy, plus six glob-coverage probes (see [Vale](#vale)) |
|
||||
| `check-scope-walkup-sync` | `validate.sh`, `validate-provenance.sh`, `new-agent.sh` and `new-skill.sh`'s four independent `$HOME`/`.git`/`apm.yml` walk-up ports still agree behaviorally |
|
||||
| `check-executables-allow-sync` | root `apm.yml`'s `executables.allow` key names kyberforge's actual version (see [apm gates](#apm-gates)) |
|
||||
|
||||
`check-executables-allow-sync` is the odd one in this group: it guards a *silent failure* rather than
|
||||
drift in generated text.
|
||||
|
||||
**Artifact validators**
|
||||
|
||||
| Hook | Guards |
|
||||
|---|---|
|
||||
| `check-apm-agents-valid` | runs agent-audit's `validate.sh` over every real `plugins/*/.apm/agents/*.agent.md` (see [Agent files](#agent-files-take-the-description-gates-not-the-body-gate)) |
|
||||
|
||||
**apm's own gates**
|
||||
|
||||
| Hook | Guards |
|
||||
|---|---|
|
||||
| `apm-marketplace-check` | every `marketplace.packages[]` entry resolves, including network reachability of remote refs |
|
||||
| `apm-audit-ci` | `apm audit --ci` once per manifest — root plus each of the six plugin packages |
|
||||
| `apm-pack-check-clean` | `apm pack --check-versions --check-clean --dry-run` — the compiled marketplace still matches what `apm.yml` + `.apm/` would generate, and per-package versions agree with the `per_package` strategy |
|
||||
|
||||
**Host validators** (both need the `claude` CLI on PATH)
|
||||
|
||||
| Hook | Guards |
|
||||
|---|---|
|
||||
| `validate-plugins` | `claude plugin validate --strict` on every plugin directory |
|
||||
| `validate-marketplace` | `claude plugin validate --strict` on the root marketplace manifest |
|
||||
|
||||
**Release**
|
||||
|
||||
| Hook | Guards |
|
||||
|---|---|
|
||||
| `check-release-needed` | on a real `git push` to `main` only — fails if files exposed via `.pre-commit-hooks.yaml` changed since the last tag. A no-op everywhere else, including under `pre-commit run --hook-stage pre-push` (see [the caveat above](#running-the-gates)) |
|
||||
|
||||
Four of these shell out to `apm`: `apm-marketplace-check`, `apm-audit-ci`, `apm-pack-check-clean`,
|
||||
and `check-plugin-content-sync` (via `scripts/sync-plugin-content.sh`, which wraps `apm pack`). The
|
||||
first and third are bare `apm …` entries and the second is a `bash -c` loop calling `apm` once per
|
||||
package, so without the CLI the push dies with an unhelpful "command not found". Install with
|
||||
`apm-install`, or `curl -sSL https://aka.ms/apm-unix | sh`; verify with `apm --version`. `jq` is
|
||||
needed by `scripts/check-manifests.sh` and `scripts/sync-plugin-content.sh` — those at least fail
|
||||
loudly (`Error: jq is required but not installed`).
|
||||
|
||||
## Skill and agent context gates (ADR-0020)
|
||||
|
||||
The `skill-size-check` pre-commit hook, scoped to `^plugins/[^/]+/\.apm/skills/[^/]+/SKILL\.md$`,
|
||||
runs `scripts/skill-size-check.sh`. That scope means it never lints the
|
||||
`plugins/kyberforge/docs/research/examples/` reference skills. It is also shipped to external repos
|
||||
as `kyberforge-skill-size-check` (see
|
||||
[External consumers](#external-consumers-the-root-pre-commit-hooksyaml)).
|
||||
|
||||
### Two independent gate families, neither replaced the other
|
||||
|
||||
**Family 1 — agentskills.io spec backstop** (unchanged, conformance not quality):
|
||||
|
||||
| Constant | Value | Measured over |
|
||||
|---|---|---|
|
||||
| `MAX_LINES` | 500 | whole file, **frontmatter included** |
|
||||
| `MAX_WORDS` | 2,770 | whole file, **frontmatter included** |
|
||||
|
||||
**Family 2 — ADR-0020 context budget** (measured differently, on purpose):
|
||||
|
||||
| Check | SUGGESTION | FAIL | Measured over |
|
||||
|---|---|---|---|
|
||||
| `description` characters | 250 | 400 | the YAML-**folded** value |
|
||||
| body words | 600 | 900 | **body only** — everything after the frontmatter's closing `---` |
|
||||
|
||||
Plus two hard FAILs with no suggestion tier:
|
||||
|
||||
- **A missing, valueless or `null` `description:`.** Not a skip. The description is the one field
|
||||
preloaded into every session, so a gate that declines to measure it reports green. (This is not
|
||||
hypothetical: `description:` with no value followed by `model: sonnet` let a line regex capture the
|
||||
*next* key, which looked non-empty, so the "missing or empty" branch never fired and every gate
|
||||
below early-returned on the genuinely empty folded value — exit 0, zero output, on a blocking gate.)
|
||||
- **Every `references/<file>.md` a body names must exist** on disk. A dispatch table pointing at a
|
||||
file that was never written is a silently dead branch, and nothing else in the gate/audit/vale
|
||||
stack notices it.
|
||||
|
||||
A file can sit well inside one family and fail the other. 2,770 whole-file words is a conformance
|
||||
backstop; 900 body-only words is a quality gate. Conflating them is what produced the current state.
|
||||
|
||||
### An unresolved routing target is not automatically a FAIL
|
||||
|
||||
A boundary-clause target that resolves to no skill or agent has **three** possible verdicts, not one
|
||||
(`unresolved_targets()` in `scripts/skill-size-check.sh`):
|
||||
|
||||
| Verdict | When |
|
||||
|---|---|
|
||||
| **SUGGESTION** — the default | the target does not resolve and neither promotion condition below holds |
|
||||
| **blocking ERROR** | the target is **terminal** (not a compound modifier) **and** either written in route notation (`/name` for any name; `-> name` only when the name is hyphenated — see the gap below) **or** corroborated by another target in the same sentence that *does* resolve |
|
||||
| **INFO, "DID NOT RUN"** | no skill universe could be determined for the path at all — the targets are named and left unchecked, exit 0 |
|
||||
|
||||
The default is deliberately soft because a hyphenated word in a boundary clause is as likely to be a
|
||||
tool, a file format or an English compound as a route: "pre-commit hooks" is prose about a tool and
|
||||
never reaches the check at all, being a compound modifier rather than a terminal name. The
|
||||
SUGGESTION text says how to opt in — write it as `/name` or `-> name` and it gets checked properly.
|
||||
|
||||
**Known gap: the arrow form only works for hyphenated names.** Target extraction is built on
|
||||
`NAME_HYPH` (`scripts/skill-size-check.sh:543`), which requires at least one hyphen, and
|
||||
`ARROW_BOUNDARY` (`:561`) inherits that. So `-> gitea-prs` is extracted and checked, while
|
||||
`-> triage` is not extracted at all — no ERROR, no SUGGESTION, exit 0. The unicode arrow `→` is not
|
||||
recognised in either case. This makes the SUGGESTION's own advice unsafe for a single-word skill:
|
||||
taking it silences the finding rather than checking it. `/name` has no such restriction and is the
|
||||
form to prefer. Tracked as a defect; `tests/test-adr0020-targets.sh` has one arrow case and its
|
||||
target happens to be hyphenated, so nothing currently covers this.
|
||||
|
||||
Corroboration is what makes the soft default safe: a sentence whose *other* target resolves is
|
||||
demonstrably a routing sentence, so a sibling that does not resolve is a typo rather than a noun, and
|
||||
gets promoted.
|
||||
|
||||
### Target resolution walk
|
||||
|
||||
Resolution walks up **from the file being checked** — never from the script's own location. Deriving
|
||||
it from `${BASH_SOURCE}` leaked holocron's 39-skill universe into every consumer repo running the
|
||||
hook through pre-commit, so a consumer skill routing to `skill-audit` resolved against a plugin it
|
||||
had never installed.
|
||||
|
||||
The walk finds an **authoring root**: the nearest ancestor holding `plugins/*/.apm/skills` or
|
||||
`plugins/*/.apm/agents`, falling back to the nearest ancestor holding `.git`. **Two passes, not one
|
||||
interleaved walk**, so a nested `.git` (a submodule, a sub-package worktree) cannot beat a real
|
||||
monorepo root further up.
|
||||
|
||||
The universe is then:
|
||||
|
||||
1. every skill and agent under `<root>/plugins/*/` — sibling plugins resolve, which is what a
|
||||
monorepo means;
|
||||
2. the checked file's own apm package;
|
||||
3. the packages that package declares in **its own** `apm.yml` `dependencies.apm`.
|
||||
|
||||
The **root** manifest's `dependencies:` block is not read, and no plugin here declares a cross-plugin
|
||||
apm dependency — none needs to.
|
||||
|
||||
Deployed `.claude/` / `.agents/` trees are consulted **only** when the walk found no plugin monorepo
|
||||
root, whether it landed on a bare `.git` ancestor or on nothing at all. That is the consumer case.
|
||||
|
||||
**The gate keys on which of the two passes matched, never on whether the root contributed a new
|
||||
name.** A name-count delta looks equivalent and is not: `_collect_authoring_root()` re-collects the
|
||||
checked file's own plugin, whose names the earlier steps already added, so a single-plugin monorepo
|
||||
shows a delta of zero and would wrongly reach for the deployed trees — including the user's global
|
||||
`~/.claude/skills`, making the verdict depend on what happens to be installed.
|
||||
|
||||
Why it matters: those trees are gitignored `apm install` output, present only on a machine that has
|
||||
run it. Four cross-plugin targets here (`gitea-branches` → `git-branches`, `gitea-branches` →
|
||||
`git-history`, `gitea-issues` → `git-branches`, `gitea-workflow` → `git-workflow`) once resolved
|
||||
through `.claude/skills/` alone, so **the same commit measured 2 dangling targets on a developer
|
||||
machine and 6 on a fresh clone**. A gate shipping hot with no baseline cannot give two answers.
|
||||
|
||||
Verified fixed: running the hook over a tree holding only `plugins/` and the root `apm.yml`, with no
|
||||
`.claude/` or `.agents/` anywhere, produces findings identical to the working tree — **26 description
|
||||
FAILs, 9 body FAILs, 2 dangling targets, 0 missing references, 58 SUGGESTIONs**.
|
||||
|
||||
### SUGGESTION-only checks
|
||||
|
||||
Three more, deterministic to measure but judgment to act on:
|
||||
|
||||
- a description with **no boundary clause at all**;
|
||||
- a `## Gotchas` section with **more than five entries**;
|
||||
- a `## Gotchas` section over **25% of the body**.
|
||||
|
||||
### `verbose: true` is load-bearing
|
||||
|
||||
The hook is declared `verbose: true` so the SUGGESTION tier is audible. pre-commit prints nothing at
|
||||
all for a passing hook, and a SUGGESTION deliberately does not fail — without verbose every
|
||||
suggestion is swallowed, which is exactly the invisibility ADR-0013 records for Vale warnings.
|
||||
ADR-0020's preload arithmetic depends on it: writing to the 400-char FAIL delivers roughly half the
|
||||
cut that writing to the 250-char SUGGESTION does, so the intended saving depends entirely on that
|
||||
tier being visible. The numbers, and the measurement method behind them, are not restated here —
|
||||
they live in ADR-0020's Consequences section, under "A ceiling does not produce an average", whose
|
||||
figures are pinned to the base commit the decision was taken on (`f9b919d`). Quoting them here would
|
||||
just create a second copy to go stale. It costs nothing on a clean file — the script prints only
|
||||
findings.
|
||||
|
||||
### Duplicated constants
|
||||
|
||||
`skill-audit`'s `validate.sh` holds a second copy of the four ADR-0020 constants
|
||||
(`DESC_SUGGEST_CHARS` / `DESC_MAX_CHARS` / `BODY_SUGGEST_WORDS` / `BODY_MAX_WORDS`), and
|
||||
`agent-audit`'s `validate.sh` holds a third copy of the two description constants. They are copied
|
||||
rather than imported because a cache-installed plugin's scripts cannot read files outside their own
|
||||
plugin directory. `tests/test-skill-size-check.sh` asserts the copies agree, so drift fails CI rather
|
||||
than silently letting an audit bless a skill the commit hook then rejects. The shared boundary
|
||||
resolver block is embedded verbatim in all three scripts between `BEGIN`/`END ADR-0020 SHARED
|
||||
BOUNDARY RESOLVER` markers and must stay byte-identical.
|
||||
|
||||
### `python3` and PyYAML are hard requirements
|
||||
|
||||
Both, and neither is a best-effort accelerator.
|
||||
|
||||
`python3` because the script measures the **folded** `description` value. Most descriptions here are
|
||||
`>`-block scalars, so a regex over the raw lines measures indentation and newlines instead of the
|
||||
value. Missing it fails the hook with an install pointer rather than skipping the ADR-0020 checks,
|
||||
which would be a vacuous green. In practice it is already present — pre-commit is itself a Python
|
||||
application.
|
||||
|
||||
**PyYAML** because the hand-rolled fallback frontmatter reader has been **removed deliberately**. It
|
||||
disagreed with a real parser across the FAIL boundary — one corpus description measured 270
|
||||
characters parsed and 412 unparsed — and a quoted `"description"` key or an explicit
|
||||
`description: null` returned empty from it, silently skipping the description *and* routing checks. A
|
||||
reader that mis-parses an unfamiliar scalar shape reports a clean pass on a file it never measured,
|
||||
which is the exact vacuous-green failure the `python3` check exists to avoid. `pip install pyyaml`
|
||||
(or `python3 -m pip install PyYAML`, or the distro's `python3-yaml`) if the hook reports it missing.
|
||||
|
||||
## Agent files take the description gates, not the body gate
|
||||
|
||||
`check-apm-agents-valid` runs agent-audit's `validate.sh` over every real
|
||||
`plugins/*/.apm/agents/*.agent.md`. It derives its expected file set from `git ls-files` — the pattern
|
||||
`tests/run-bats.sh` established — so an agent file deleted from the worktree but still tracked fails
|
||||
the run, and **discovering zero agent files is an error, not a pass**. An untracked *new* agent file
|
||||
is still validated: the derivation is one-directional on purpose, so uncommitted work is not blocked
|
||||
but also cannot bypass the gate.
|
||||
|
||||
The hook exists because `validate.sh` was previously exercised only by `check-scope-walkup-sync`,
|
||||
against synthetic `mktemp` fixtures — it had never run against the agent files it governs. That is
|
||||
how ADR-0016 could be amended to bless a `disallowedTools` frontmatter field while `validate.sh`'s
|
||||
allowlist still rejected it: spec and enforcer disagreed and every gate stayed green.
|
||||
|
||||
Agents take the ADR-0020 **description** gates (agent-audit's `validate.sh` holds its own copy of
|
||||
those two constants) and, deliberately, **no body word gate**. A skill body is loaded into the
|
||||
caller's context and competes with the live conversation; an agent body becomes the system prompt of
|
||||
a *fresh* context. The rationale for the 900-word FAIL does not transfer. A bats test pins that
|
||||
absence in agent-audit's validator — adding a body gate there contradicts the ADR rather than fixing
|
||||
an inconsistency.
|
||||
|
||||
**Be precise about the scope of that guarantee: it holds for the *validator*, not for the shared
|
||||
script.** `scripts/skill-size-check.sh` applies its body gate to whatever path it is handed, and
|
||||
|
||||
```
|
||||
bash scripts/skill-size-check.sh plugins/*/.apm/agents/*.agent.md
|
||||
```
|
||||
|
||||
exits 1 today with 900-word body FAILs on `git-orchestrate` (933), `gitea-orchestrate` (1,199) and
|
||||
`apm-orchestrate` (1,080). Agent files escape only because the hook definitions filter on `SKILL.md`
|
||||
— a file-pattern accident that happens to implement the design, not the design itself. **Do not
|
||||
"extend" that hook's `files:` pattern to cover agents** on the assumption that the script already
|
||||
knows the difference; doing so silently enforces a gate ADR-0020 declines to set.
|
||||
|
||||
## Current retrofit status
|
||||
|
||||
**The ADR-0020 gates ship hot, with no baseline file.** A shrinking baseline recording each
|
||||
non-compliant skill's current numbers was considered and rejected in favour of hot gates.
|
||||
|
||||
Two independent hot gates are currently red, and the first will not warn you about the second.
|
||||
|
||||
| Gate | Current findings |
|
||||
|---|---|
|
||||
| `skill-size-check` | **26 of 39** descriptions and **9 of 39** bodies exceed their FAIL tier; 2 dangling targets; 58 SUGGESTIONs |
|
||||
| `Kyberforge.CompositionNote` (Vale) | **10 errors across four skills**: `gitea-issues`, `gitea-labels-milestones`, `gitea-prs`, `gitea-workflow` |
|
||||
|
||||
`Kyberforge.CompositionNote` is the ADR-0020 Vale rule banning composition and architecture prose
|
||||
from a description. Every Vale rule here is `level: error` with no ignorable tier, so touching any of
|
||||
those four skills means fixing its prose findings as well as its size findings.
|
||||
|
||||
Consequence: editing a non-compliant skill *for any reason* means retrofitting it to the contract
|
||||
first — a one-line fix to `gitea-prs` cannot be committed until that skill complies. This is
|
||||
deliberate; it guarantees convergence and avoids a half-state. Tracked as Gitea issue **#99**.
|
||||
|
||||
Check where a skill stands before starting, and check **both** gates:
|
||||
|
||||
```
|
||||
pre-commit run skill-size-check --all-files # size/context only
|
||||
pre-commit run --all-files # size AND Vale
|
||||
```
|
||||
|
||||
Scoping a retrofit off `skill-size-check` output alone leaves you blocked at the second gate.
|
||||
|
||||
## Vale
|
||||
|
||||
Install the `vale` binary — `brew install vale` (macOS), `snap install vale` (Linux),
|
||||
`choco install vale` (Windows), or see <https://vale.sh/docs/vale-cli/installation/>. No `vale sync`
|
||||
is needed: the `Kyberforge` styles are **committed** under
|
||||
`plugins/kyberforge/.apm/skills/{skill-audit,agent-audit}/assets/vale/styles/`, not downloaded
|
||||
packages (ADR-0014).
|
||||
|
||||
### Two copies, one canonical
|
||||
|
||||
Wiring Vale as a deterministic prefilter for `skill-audit`/`agent-audit`'s Description dimension
|
||||
(motivation: issue #84) is repo-specific, not part of the generic `lint` plugin, so it does not live
|
||||
in `plugins/lint/` — and per ADR-0014 it no longer lives at the repo root either. It lives **twice**,
|
||||
one copy per skill, both under `plugins/kyberforge/.apm/skills/`:
|
||||
|
||||
| Copy | Styles | `.vale.ini` sections |
|
||||
|---|---|---|
|
||||
| `agent-audit/assets/vale/` — **canonical** | `Kyberforge`, `KyberforgeCopilot` | `[**/agents/*.md]`, `[**/*.agent.md]` |
|
||||
| `skill-audit/assets/vale/` — smaller duplicate | `Kyberforge` | `[**/SKILL.md]` |
|
||||
|
||||
Duplicated rather than shared because a plugin's cache-install copies only each skill's own files —
|
||||
there is no cross-skill sharing to point at. `check-vale-style-sync` at pre-push is what keeps them
|
||||
from drifting; `KyberforgeCopilot` is the one deliberate inequality, being scoped only to `.agent.md`
|
||||
files for the Copilot-only "`Use proactively` has no effect" check.
|
||||
|
||||
### What Vale owns, and what stays LLM judgment
|
||||
|
||||
Eleven rule files across the two copies, six distinct rules:
|
||||
|
||||
| Rule | Vale scope | Bans | From |
|
||||
|---|---|---|---|
|
||||
| `Kyberforge.DescriptionOpener` | `text.frontmatter.description` | non-imperative openers ("This skill/agent…") | issue #84 |
|
||||
| `Kyberforge.VagueWording` | `text.frontmatter.description` | vague capability wording ("helps with", "utilize", …) | issue #84 |
|
||||
| `Kyberforge.PaddingPhrase` | `text` | generic "see `references/` for details" padding | issue #84 |
|
||||
| `KyberforgeCopilot.ProactivePhrase` | `text.frontmatter.description` | `Use proactively` (no effect in Copilot) | issue #84 |
|
||||
| `Kyberforge.SentenceOpenerThereIs` | `sentence` | "There is/are" sentence openers | ADR-0013 |
|
||||
| `Kyberforge.CompositionNote` | `text.frontmatter.description` | architecture and composition prose in a description | ADR-0020 |
|
||||
|
||||
Vale covers the **pattern-matchable** sub-checks named in issue #84 plus, per ADR-0013, one
|
||||
cherry-picked body-wide prose-pattern rule. Everything else stays LLM judgment: defaults-vs-menus,
|
||||
why-rationale, the non-pattern-matchable body-discipline calls, near-miss exclusion strength, and
|
||||
control calibration. New rules land directly in `styles/Kyberforge` and block immediately — there is
|
||||
no trial tier.
|
||||
|
||||
The cherry-pick record, so it is not re-litigated:
|
||||
|
||||
- `Kyberforge.SentenceOpenerThereIs` **landed** — 22 held-out hits, both in-corpus hits clean
|
||||
rewrites, zero suppressions needed.
|
||||
- `Kyberforge.VagueQualifier` was cherry-picked and then **deleted**. 2 hits across the corpus as it
|
||||
stood on 2026-08-08 (before the `.apm/` restructure): one marginal, and one unfixable false
|
||||
positive — `caveman/SKILL.md` quotes `of course` as an example of filler, a mention rather than a
|
||||
use — which forced the repo's only Vale suppression comments.
|
||||
- `governance.md` and `CONTROLS.md` were evaluated as rule sources and **excluded**: nothing
|
||||
prose-pattern-matchable to mine.
|
||||
|
||||
### Why every rule is `level: error`
|
||||
|
||||
Every alert is a FAIL, with no ignorable tier — same all-or-nothing model as shellcheck, the test
|
||||
suite, and conventional-pre-commit. Graded severities do not work here: **Vale's exit code keys on
|
||||
`error` alerts alone**, so a `warning` or `suggestion` rule exits 0, and pre-commit swallows a
|
||||
passing hook's output. Such a rule would be invisible and would block nothing.
|
||||
|
||||
`MinAlertLevel` and `--minAlertLevel` are correspondingly **absent** from both `.vale.ini` files and
|
||||
from the hook definitions. Under this model they are no-ops; adding one is not a missing knob.
|
||||
|
||||
The `verbose: true` escape hatch that makes `skill-size-check`'s SUGGESTION tier audible has no
|
||||
analogue here — Vale has no tier to make audible.
|
||||
|
||||
### External consumers: the root `.pre-commit-hooks.yaml`
|
||||
|
||||
The root `.pre-commit-hooks.yaml` exposes both Vale copies (`kyberforge-vale-audit-skill`,
|
||||
`kyberforge-vale-audit-agent`) plus `kyberforge-skill-size-check`, so any external repo can enforce
|
||||
the same rules with `repo: <this-repo-url>, rev: <tag>` in its own `.pre-commit-config.yaml`.
|
||||
pre-commit clones the pinned rev into its own cache, independent of whether Claude Code or the
|
||||
`kyberforge` plugin is installed at all; the same mechanism covers CI via `pre-commit run
|
||||
--all-files`. `skill-size-check` has no external asset dependency, so it needed no relocation under
|
||||
ADR-0014 — only exposure.
|
||||
|
||||
This repo's own `vale-audit-prefilter-skill` / `-agent` hooks consume the **identical**
|
||||
plugin-bundled copies via `repo: local`. Deliberately not a third root copy, and deliberately **not a
|
||||
pinned self-reference** — a pinned self-reference would lint working-tree edits against the last
|
||||
tagged release rather than against the change being made.
|
||||
|
||||
### Pre-commit
|
||||
|
||||
Two prefilter hooks, with `.apm/`-scoped `files:` patterns:
|
||||
|
||||
| Hook | Pattern |
|
||||
|---|---|
|
||||
| `vale-audit-prefilter-skill` | `^plugins/[^/]+/\.apm/skills/[^/]+/SKILL\.md$` |
|
||||
| `vale-audit-prefilter-agent` | `^plugins/[^/]+/\.apm/agents/[^/]+\.agent\.md$` |
|
||||
|
||||
Only the **authoring source** triggers them. A `SKILL.md` in the generated flat mirror matches
|
||||
neither pattern, so prose findings surface only when you edit the file you are supposed to be
|
||||
editing. Without the binary the hooks fail with a bare "command not found" and no install pointer.
|
||||
|
||||
**Two hooks, not one combined hook.** Both manifests split the prefilter in two precisely because a
|
||||
single hook can point at only one copy, and that copy would silently 0-file-skip the other file
|
||||
shape (see [A 0-file Vale run is NOT RUN](#a-0-file-vale-run-is-not-run)).
|
||||
|
||||
### The `.vale.ini` globs do no scoping
|
||||
|
||||
Each `.vale.ini`'s section globs are **path-agnostic** — `[**/SKILL.md]` for skill-audit's copy,
|
||||
`[**/agents/*.md]` and `[**/*.agent.md]` for agent-audit's — and constrain filename *shape*, not
|
||||
location: Vale's `*` crosses `/`. A `SKILL.md` outside `plugins/` (a project-scope
|
||||
`.claude/skills/foo/SKILL.md`, say) still matches `[**/SKILL.md]` and gets linted normally.
|
||||
|
||||
All scoping therefore comes from the pre-commit hook's own `files:` regex and from the audit skills
|
||||
passing one explicit file per invocation. The two manifests scope **differently on purpose**:
|
||||
|
||||
| Manifest | `-skill` | `-agent` |
|
||||
|---|---|---|
|
||||
| `.pre-commit-config.yaml` (pins this repo's layout) | `^plugins/[^/]+/\.apm/skills/[^/]+/SKILL\.md$` | `^plugins/[^/]+/\.apm/agents/[^/]+\.agent\.md$` |
|
||||
| `.pre-commit-hooks.yaml` (layout-agnostic for consumers) | `(^\|/)SKILL\.md$` | `(^\|/)agents/[^/]+\.md$\|\.agent\.md$` |
|
||||
|
||||
Narrowing a `.vale.ini` glob to a `plugins/`-shaped path to "tighten" it breaks the consumer case,
|
||||
and `check-vale-style-sync`'s probe set is built to catch exactly that.
|
||||
|
||||
### `vale-wrap.sh`, never bare `vale`
|
||||
|
||||
Both audit skills' Step 1 and both pre-commit hooks call **each copy's own**
|
||||
`scripts/vale-wrap.sh`, not `vale`. It works around a confirmed **Vale 3.15.2** limitation:
|
||||
`text.frontmatter.description` silently stops matching on most — not all — multi-line descriptions.
|
||||
|
||||
Verified by reproduction on a deliberately-bad fixture, not assumed:
|
||||
|
||||
| Description scalar spanning 2+ lines | Vale's behaviour |
|
||||
|---|---|
|
||||
| `>` folded block | 0 alerts, exit 0 — **broken** |
|
||||
| plain (unquoted) continuation lines | 0 alerts, exit 0 — **broken** |
|
||||
| single- or double-quoted, wrapped | 0 alerts, exit 0 — **broken** |
|
||||
| `\|` literal block | alerts fire, exit 1 — lints normally |
|
||||
|
||||
The wrapper flattens the three broken forms to a single-line scalar in a scratch copy — or, for the
|
||||
rare value no inline scalar can spell verbatim, a `|-` block with one content line — padding with
|
||||
blank lines so **every other line number is unchanged**. `|` literal blocks and single-line
|
||||
descriptions pass through untouched. Most descriptions in this repo are `>` blocks, so before the
|
||||
wrapper a bad description in any of the three broken forms sailed straight through the prefilter.
|
||||
|
||||
### The `--config` argv defect
|
||||
|
||||
Handed **no `--config` at all**, the wrapper falls back to its own sibling `assets/vale/.vale.ini`,
|
||||
located from `${BASH_SOURCE[0]}` rather than from the cwd. That is why both manifests' `entry:` is
|
||||
now the bare script path with **no argument after it**.
|
||||
|
||||
pre-commit prefixes only `entry[0]` with the hook-repo clone path (`cmd = (prefix.path(cmd[0]),
|
||||
*cmd[1:])`), so every later argument resolves against the **consuming** repo's root. A `--config` in
|
||||
`.pre-commit-hooks.yaml` therefore pointed at a path no consumer has and hard-failed every external
|
||||
run with `E100 [--config] Runtime error`.
|
||||
|
||||
`.pre-commit-config.yaml` drops the argument too, deliberately keeping the two entries identical.
|
||||
The local `repo: local` hook resolved its `--config` correctly only because the consuming repo *was*
|
||||
this repo — and that divergence is why three review rounds exercised a path no external consumer
|
||||
takes and missed the defect. **Do not reintroduce a `--config` to either manifest to make the local
|
||||
run "explicit".**
|
||||
|
||||
An explicit `--config` from any other caller still wins, in all three argv forms (`--config X`,
|
||||
`--config=/abs`, `--config=rel`), and a relative one resolves against the caller's cwd — matching
|
||||
bare `vale`, not the repo root.
|
||||
|
||||
Both audit skills' Step 1 passes no `--config` either. Step 1 resolves the script relative to the
|
||||
skill's own directory so the call works from an installed plugin cache; a relative `--config`
|
||||
alongside it would resolve against the cwd instead, yielding `E100 Runtime error … does not exist`
|
||||
and exit 2 — which both skills' fallback misreads as "vale unavailable" and silently downgrades to
|
||||
full LLM judgment.
|
||||
|
||||
`tests/test-vale-wrap.sh` regression-tests this against **skill-audit's** copy specifically: its
|
||||
fixtures are all `SKILL.md`-shaped, and only skill-audit's `.vale.ini` carries that glob section.
|
||||
|
||||
### A 0-file Vale run is NOT RUN
|
||||
|
||||
Vale reports 0 files only when the path it is handed matches **no glob section at all** — a
|
||||
differently-named file, or a directory argument holding nothing that matches. That run prints
|
||||
|
||||
```
|
||||
✔ 0 errors ... in 0 files.
|
||||
```
|
||||
|
||||
and exits 0, indistinguishable from a clean pass. Both audits therefore treat a 0-file Vale run as
|
||||
**NOT RUN** and fall back to full LLM judgment rather than reporting the Description dimension
|
||||
clean.
|
||||
|
||||
### Pre-push
|
||||
|
||||
`vale` is a **pre-push** dependency too, not only pre-commit. `check-vale-style-sync` runs **six
|
||||
glob-coverage probes** by invoking `vale --config` — one representative path per file shape the
|
||||
prefilter is supposed to cover. They are the only assertions in the script that catch a `.vale.ini`
|
||||
glob typo (`[**/SKILL.md]` → `[**/SKILLS.md]`), the failure mode where every text-level check stays
|
||||
clean while vale lints zero files. As a warning this self-disabled on exactly that mutation and
|
||||
exited 0, and since pre-commit swallows a passing hook's output the stderr line was never seen — the
|
||||
hook reported `Passed`. Missing `vale` is therefore a hard failure here.
|
||||
|
||||
The opt-out is `CHECK_VALE_STYLE_SYNC_ALLOW_MISSING_VALE=1`, and **it is not `SKIP=`**: the hook
|
||||
still runs and still asserts everything verifiable from file text, but the six probes do not, and its
|
||||
summary says so explicitly —
|
||||
|
||||
```
|
||||
Vale style sync check passed (text-level only, vale unavailable): … 0 glob probe(s) verified.
|
||||
```
|
||||
|
||||
Use it only on a machine that genuinely cannot install `vale`, and read that line as "the glob axis
|
||||
was not checked", not as a pass. The hook is `verbose: true` for exactly that reason — its clean
|
||||
output is a single line, so it costs one line per push.
|
||||
|
||||
### Mentioning banned phrasing without tripping the rule
|
||||
|
||||
House convention: banned phrasing that must be **mentioned** rather than used goes in backticks or a
|
||||
fenced code block. Vale skips code spans and fences, so no suppression is needed — which is why this
|
||||
document quotes `Use proactively` and "There is/are" the way it does.
|
||||
|
||||
Inline `<!-- vale Rule = NO -->` is the fallback **only** where backticking is impossible. Use the
|
||||
HTML-comment form; the MDX `{/* */}` form does not work in plain Markdown. The one time a rule forced
|
||||
suppression comments, the rule was deleted instead (see the `VagueQualifier` entry above).
|
||||
|
||||
## Tests
|
||||
|
||||
```
|
||||
bash tests/run-tests.sh # every test-*.sh plus the bats suite
|
||||
bash tests/run-tests.sh --bats-only # just bats
|
||||
```
|
||||
|
||||
First run auto-initializes the bats submodules; no manual `git submodule update` needed.
|
||||
|
||||
**Exit 77 = SKIPPED.** A suite that skips because a dependency is missing does **not** fail an ad-hoc
|
||||
run. The pre-push hook invokes the same script as `--strict` (`RUN_TESTS_STRICT=1` is equivalent),
|
||||
where a skip **does** fail the push: at pre-push a skip means one of the documented dependencies is
|
||||
absent on this machine, so the gate would otherwise report success having run fewer suites than it
|
||||
appears to. Without `--strict` the gate once went green having verified 15 of 17 suites on a
|
||||
vale-less PATH, with the skip list swallowed. Without vale, three suites skip —
|
||||
`test-check-vale-style-sync.sh`, `test-vale-hooks-consumer.sh`, `test-vale-wrap.sh` — and the strict
|
||||
failure names each one and what to install.
|
||||
|
||||
`tests/run-bats.sh` derives the set of `.bats` files it expects from `git ls-files`, so a `.bats`
|
||||
file deleted from the worktree but still tracked in the index fails the run rather than silently
|
||||
shrinking the suite. Remove one with `git rm` (or stage the deletion) when intentional; an untracked
|
||||
new `.bats` file is picked up and needs no ceremony.
|
||||
|
||||
Both discovery walks (`tests/run-bats.sh` and `tests/run-tests.sh`) exclude `apm_modules/`:
|
||||
`apm install` materializes a full copy of every plugin there, and running a dependency's copy of a
|
||||
`.bats` file breaks its relative path to the bats helpers — **167 spurious failures** before the
|
||||
exclusion landed.
|
||||
|
||||
## apm gates
|
||||
|
||||
### `apm-audit-ci`
|
||||
|
||||
Runs `apm audit --ci` **once per manifest** — the root one and each of the six plugin packages —
|
||||
because the root-only invocation audits the marketplace manifest and **nothing else**, and
|
||||
`apm-pack-check-clean` does not parse plugin `dependencies:` blocks either. Verified: a malformed
|
||||
dependency entry passes `apm pack --check-versions --check-clean --dry-run` and fails
|
||||
`apm audit --ci` in that package's directory. Costs ~0.5s per package.
|
||||
|
||||
It verifies **exactly two things** per manifest and claims no more:
|
||||
|
||||
- **manifest-parse** — each `apm.yml` parses as a valid APM manifest. Unconditional; verified to fire
|
||||
on a dependency entry missing its `git`/`path`/`registry` field (`Cannot parse apm.yml`).
|
||||
- **lockfile-exists** — any package declaring dependencies has a consistent `apm.lock.yaml`.
|
||||
Conditional, and vacuous while every plugin `apm.yml` declares `dependencies: {apm: [], mcp: []}`;
|
||||
it arms itself the moment one does not (verified by adding a git dependency to
|
||||
`plugins/lint/apm.yml`).
|
||||
|
||||
It does **not** enforce an org policy. apm discovers one from the git remote and only understands
|
||||
github.com and Azure DevOps, so against this repo's self-hosted Gitea remote it prints:
|
||||
|
||||
```
|
||||
No org policy found at unknown; enforcement skipped
|
||||
```
|
||||
|
||||
**Do not "fix" that with `policy.fetch_failure_default: block` in `apm.yml`.** apm's own message
|
||||
suggests it; it was tried on a scratch copy and **rejected**. With no reachable policy source it does
|
||||
not make the check meaningful, it makes it permanently red — `apm audit --ci` exits 1 with
|
||||
`No org policy found at unknown (policy.fetch_failure_default=block)` on every push, forever. A gate
|
||||
that can never go green is not a gate. Revisit only if this repo gains a policy source apm can reach.
|
||||
|
||||
It also does not scan for hidden Unicode: that scan is plain `apm audit`, a different mode (`--ci`
|
||||
refuses to combine with `--file`/`--strip`/`--dry-run`/`PACKAGE`), and plain `apm audit` here reports
|
||||
`No apm.lock.yaml found -- nothing to scan` and exits 0. Adding it would buy a second vacuous check.
|
||||
|
||||
### `check-executables-allow-sync`
|
||||
|
||||
apm gates a package's `hooks/` and `bin/` on an **exact `<package>#<version>` dictionary lookup** in
|
||||
root `apm.yml`'s `executables.allow` (`apm_cli/security/executables.py`, `is_package_approved`).
|
||||
There is no wildcard and no version-less form.
|
||||
|
||||
So bumping `plugins/kyberforge/apm.yml`'s `version:` without bumping the key **errors nowhere**: the
|
||||
entry simply stops matching, the gate blocks the hook, kyberforge's `SessionStart` hook stops
|
||||
deploying, and the apm install goes quietly stale — the exact failure ADR-0019 exists to end,
|
||||
reintroduced through the mechanism meant to secure it. ADR-0019 records this as a live failure mode;
|
||||
the release that shipped the hook hit it immediately.
|
||||
|
||||
`scripts/check-executables-allow-sync.sh` parses `version:` out of `plugins/kyberforge/apm.yml` and
|
||||
asserts root `apm.yml` carries the matching `kyberforge#<version>` key. A comment in the
|
||||
`executables:` block stays as the human-facing pointer; the hook is what actually holds. It parses
|
||||
with PyYAML where importable and falls back to a two-shape scan otherwise, so a missing pip package
|
||||
cannot become the thing that blocks every push.
|
||||
|
||||
## `.claude/settings.json`
|
||||
|
||||
**apm owns this file. Nothing repo-authored goes in it.**
|
||||
|
||||
`apm audit --ci` replays the install into a scratch tree and diffs the result byte-for-byte, so
|
||||
anything apm would not have written there — an `enabledPlugins` block, a real `hooks` entry — is
|
||||
permanent drift that fails `apm-audit-ci`. A hook you want in this repo is authored in
|
||||
`plugins/<name>/.apm/hooks/` and deployed by apm, never hand-written here.
|
||||
|
||||
Its committed content is whatever apm last wrote, which today is the merged `SessionStart` entry for
|
||||
kyberforge's `check-apm-current.sh`. That is apm's own output and it belongs in the commit (ADR-0019;
|
||||
ADR-0018's statement that the committed content is exactly `{"hooks": {}}` is superseded on that
|
||||
point only). Machine-specific settings go in the gitignored `.claude/settings.local.json`, which apm
|
||||
does not deploy and the replay does not compare; shared enforcement belongs in
|
||||
`.pre-commit-config.yaml`.
|
||||
|
||||
### Why it is excluded from `pretty-format-json`
|
||||
|
||||
It is the **sixth and last alternation** in that hook's `exclude:` pattern, and the only one there
|
||||
for a reason other than "generated manifest". Mind which number you are quoting: **six alternations,
|
||||
expanding to sixteen real files** — 3 root marketplace manifests, 2 per plugin × 6 plugins, plus this
|
||||
one.
|
||||
|
||||
`pretty-format-json --autofix` sorts object keys unless `--no-sort-keys` is passed, while apm's hook
|
||||
integrator emits insertion order (`matcher` before `hooks`, `type` before `command`). Leaving the
|
||||
file in that hook's scope therefore rewrites apm's output into a form apm would never produce on the
|
||||
way into **every** commit, and `apm-audit-ci` then reports permanent drift on a file with an empty
|
||||
`git diff` — exactly what happened when the `SessionStart` hook first landed in `2e395a4`. Re-running
|
||||
`apm install` fixes the file; leaving it in scope would re-break it on the very commit carrying the
|
||||
fix.
|
||||
|
||||
**Load-bearing. Do not tidy it out of that list** (see `LESSONS.md`, 2026-08-14).
|
||||
|
||||
## Pushing without a network
|
||||
|
||||
Exactly **two** pre-push hooks need the network, for one shared reason: root `apm.yml`'s
|
||||
`marketplace.packages[]` contains exactly one remote entry — `mattpocock-skills`,
|
||||
`source: mattpocock/skills` — and resolving it needs a `git ls-remote`.
|
||||
|
||||
| Hook | Offline failure |
|
||||
|---|---|
|
||||
| `apm-marketplace-check` (`always_run`, resolves every entry) | `No cached refs (offline)` |
|
||||
| `apm-pack-check-clean` (re-resolves the same entry) | `Error: Git network timeout during ls-remote` |
|
||||
|
||||
Pinning the entry to an exact version does **not** remove the call — an exact pin still ls-remotes.
|
||||
`--offline` rescues neither.
|
||||
|
||||
To push without a network, skip both using pre-commit's own mechanism:
|
||||
|
||||
```
|
||||
SKIP=apm-marketplace-check,apm-pack-check-clean git push
|
||||
```
|
||||
|
||||
**Skip those two alone.** Verified under `unshare -rn`: the other twelve pre-push hooks pass offline
|
||||
because they are real local checks. (`check-executables-allow-sync` landed after that run, but reads
|
||||
two local manifests and makes no network call.) Adding any other hook to `SKIP` disarms it silently.
|
||||
|
||||
`apm-audit-ci` calls `apm` too but stays local: its org-policy discovery resolves nothing on this
|
||||
remote *before* any network call, so it does not join the pair above.
|
||||
|
||||
---
|
||||
|
||||
## See also
|
||||
|
||||
- `docs/adr/0020-skill-description-and-body-context-contract.md` — the context contract, its
|
||||
enforcement table (deterministic vs. auditor judgment), and every rejected alternative
|
||||
- `docs/adr/0019-session-start-hook-keeps-the-apm-install-current.md` — the `SessionStart` hook, the
|
||||
executable-trust gate, and the version-pinned allow key
|
||||
- `docs/adr/0017-plugin-content-mirror-bridges-apm-to-host-discovery.md`,
|
||||
`docs/adr/0015-apm-replaces-plugin-marketplace-authoring.md`,
|
||||
`docs/adr/0014-vale-prefilter-ships-from-the-plugin.md` — plugin content sync, apm-generated
|
||||
manifests, committed Vale styles
|
||||
- `docs/spec/architecture.md` — directory structure, install pipeline, what is generated and what is
|
||||
hand-authored
|
||||
- `.pre-commit-config.yaml` — the hooks themselves, with inline rationale comments
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "bin",
|
||||
"version": "1.1.1",
|
||||
"description": "A place for things to be binned",
|
||||
"version": "1.1.5",
|
||||
"description": "Skills for everyday AI-assisted development work that is not tied to a single tool, forge or language, and has not yet been split into a focused plugin.",
|
||||
"author": {
|
||||
"name": "Defame1297",
|
||||
"email": "defame1297@rkdr.net",
|
||||
@@ -20,7 +20,7 @@
|
||||
"mcpServers": {
|
||||
"obsidian": {
|
||||
"args": [
|
||||
"@bitbonsai/mcpvault@latest",
|
||||
"@bitbonsai/mcpvault@0.15.0",
|
||||
"docs/"
|
||||
],
|
||||
"command": "npx",
|
||||
|
||||
15
plugins/bin/.github/plugin/plugin.json
vendored
15
plugins/bin/.github/plugin/plugin.json
vendored
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "bin",
|
||||
"version": "1.1.1",
|
||||
"description": "A place for things to be binned",
|
||||
"version": "1.1.5",
|
||||
"description": "Skills for everyday AI-assisted development work that is not tied to a single tool, forge or language, and has not yet been split into a focused plugin.",
|
||||
"author": {
|
||||
"name": "Defame1297",
|
||||
"email": "defame1297@rkdr.net",
|
||||
@@ -17,14 +17,5 @@
|
||||
"tdd",
|
||||
"research"
|
||||
],
|
||||
"mcpServers": {
|
||||
"obsidian": {
|
||||
"args": [
|
||||
"@bitbonsai/mcpvault@latest",
|
||||
"docs/"
|
||||
],
|
||||
"command": "npx",
|
||||
"type": "stdio"
|
||||
}
|
||||
}
|
||||
"mcpServers": ".mcp.json"
|
||||
}
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
"mcpServers": {
|
||||
"obsidian": {
|
||||
"args": [
|
||||
"@bitbonsai/mcpvault@latest",
|
||||
"@bitbonsai/mcpvault@0.15.0",
|
||||
"docs/"
|
||||
],
|
||||
"command": "npx",
|
||||
|
||||
@@ -33,10 +33,12 @@ copilot plugin install ./plugins/bin
|
||||
| Component | Path | Description |
|
||||
|---|---|---|
|
||||
| Skills | `.apm/skills/` → `skills/` | Slash commands available after install |
|
||||
| MCP servers | `.mcp.json` | The `obsidian` server (`npx @bitbonsai/mcpvault@latest docs/`), hand-authored at the plugin root and reinjected into both compiled `plugin.json` manifests |
|
||||
| MCP servers | `.mcp.json` | The `obsidian` server (`npx @bitbonsai/mcpvault@0.15.0 docs/`), hand-authored at the plugin root |
|
||||
|
||||
`.apm/` is the authoring source; `skills/` is the generated mirror plugin hosts scan (ADR-0017). This plugin ships no agents. It is the only plugin here with a non-empty `.mcp.json`, which is why its compiled manifests are the only ones carrying an `mcpServers` block.
|
||||
|
||||
The two compiled manifests get that block by different routes. `.claude-plugin/plugin.json` gets it from apm itself: `build_plugin_manifest`'s Claude branch calls `collect_mcp_servers`, which reads `.mcp.json`, sanitizes it, and inlines the resulting server objects. `.github/plugin/plugin.json` gets nothing from apm — the Copilot branch drops the field — so `scripts/sync-plugin-content.sh`'s `reinject_mcp_servers()` puts it back, as the **string `".mcp.json"`** rather than the resolved objects. Copilot's manifest schema types the field as "string or object — MCP server config path or inline definitions", and a path reference cannot carry a credential into a committed manifest. See ADR-0017's `mcpServers` amendment.
|
||||
|
||||
## Author
|
||||
|
||||
Defame1297
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
name: bin
|
||||
version: 1.1.1
|
||||
description: A place for things to be binned
|
||||
version: 1.1.5
|
||||
description: Skills for everyday AI-assisted development work that is not tied to a single tool, forge or language, and has not yet been split into a focused plugin.
|
||||
author:
|
||||
name: Defame1297
|
||||
email: defame1297@rkdr.net
|
||||
|
||||
@@ -24,7 +24,12 @@ Provide the path to the repo root to audit when invoking.
|
||||
| `scripts/validate-drift.sh` | Resolves referenced npm/make commands and file paths against the repo |
|
||||
| `references/sources.md` | Provenance record — sources that informed this skill and which files each contributed to |
|
||||
| `scripts/README.md` | Directory documentation for `scripts/` |
|
||||
| `tests/README.md` | Bats test dependency and run instructions |
|
||||
| `tests/validate-secrets.bats` | Bats test suite for `scripts/validate-secrets.sh` |
|
||||
| `tests/validate-structure.bats` | Bats test suite for `scripts/validate-structure.sh` |
|
||||
| `tests/validate-drift.bats` | Bats test suite for `scripts/validate-drift.sh` |
|
||||
| `tests/README.md` | (source-only) Bats test dependency and run instructions |
|
||||
| `tests/validate-secrets.bats` | (source-only) Bats test suite for `scripts/validate-secrets.sh` |
|
||||
| `tests/validate-structure.bats` | (source-only) Bats test suite for `scripts/validate-structure.sh` |
|
||||
| `tests/validate-drift.bats` | (source-only) Bats test suite for `scripts/validate-drift.sh` |
|
||||
|
||||
Rows marked **(source-only)** exist in the authoring source (`.apm/skills/agentsmd-audit/`) but are
|
||||
not present in an installed plugin: `scripts/sync-plugin-content.sh` strips `<category>/<name>/tests`
|
||||
when it generates the flat mirror, because these are dev-time fixtures no plugin host needs to
|
||||
discover (ADR-0017). Run them from a repo checkout, not from an install.
|
||||
|
||||
@@ -26,5 +26,10 @@ Provide the path to the provider-specific file to convert (and the target repo r
|
||||
| `references/sources.md` | Provenance record — the in-repo ADR precedent this skill's design is modeled on |
|
||||
| `scripts/validate-adapter.sh` | Self-check gate: reference to AGENTS.md present, no excessive duplication, adapter stays thin |
|
||||
| `scripts/README.md` | Directory documentation for `scripts/` |
|
||||
| `tests/README.md` | Bats test dependency and run instructions |
|
||||
| `tests/validate-adapter.bats` | Bats test suite for `scripts/validate-adapter.sh` |
|
||||
| `tests/README.md` | (source-only) Bats test dependency and run instructions |
|
||||
| `tests/validate-adapter.bats` | (source-only) Bats test suite for `scripts/validate-adapter.sh` |
|
||||
|
||||
Rows marked **(source-only)** exist in the authoring source (`.apm/skills/provider-adapter-author/`)
|
||||
but are not present in an installed plugin: `scripts/sync-plugin-content.sh` strips
|
||||
`<category>/<name>/tests` when it generates the flat mirror, because these are dev-time fixtures no
|
||||
plugin host needs to discover (ADR-0017). Run them from a repo checkout, not from an install.
|
||||
|
||||
@@ -24,7 +24,12 @@ Provide the path to the repo root to audit when invoking.
|
||||
| `scripts/validate-drift.sh` | Resolves referenced npm/make commands and file paths against the repo |
|
||||
| `references/sources.md` | Provenance record — sources that informed this skill and which files each contributed to |
|
||||
| `scripts/README.md` | Directory documentation for `scripts/` |
|
||||
| `tests/README.md` | Bats test dependency and run instructions |
|
||||
| `tests/validate-secrets.bats` | Bats test suite for `scripts/validate-secrets.sh` |
|
||||
| `tests/validate-structure.bats` | Bats test suite for `scripts/validate-structure.sh` |
|
||||
| `tests/validate-drift.bats` | Bats test suite for `scripts/validate-drift.sh` |
|
||||
| `tests/README.md` | (source-only) Bats test dependency and run instructions |
|
||||
| `tests/validate-secrets.bats` | (source-only) Bats test suite for `scripts/validate-secrets.sh` |
|
||||
| `tests/validate-structure.bats` | (source-only) Bats test suite for `scripts/validate-structure.sh` |
|
||||
| `tests/validate-drift.bats` | (source-only) Bats test suite for `scripts/validate-drift.sh` |
|
||||
|
||||
Rows marked **(source-only)** exist in the authoring source (`.apm/skills/agentsmd-audit/`) but are
|
||||
not present in an installed plugin: `scripts/sync-plugin-content.sh` strips `<category>/<name>/tests`
|
||||
when it generates the flat mirror, because these are dev-time fixtures no plugin host needs to
|
||||
discover (ADR-0017). Run them from a repo checkout, not from an install.
|
||||
|
||||
@@ -26,5 +26,10 @@ Provide the path to the provider-specific file to convert (and the target repo r
|
||||
| `references/sources.md` | Provenance record — the in-repo ADR precedent this skill's design is modeled on |
|
||||
| `scripts/validate-adapter.sh` | Self-check gate: reference to AGENTS.md present, no excessive duplication, adapter stays thin |
|
||||
| `scripts/README.md` | Directory documentation for `scripts/` |
|
||||
| `tests/README.md` | Bats test dependency and run instructions |
|
||||
| `tests/validate-adapter.bats` | Bats test suite for `scripts/validate-adapter.sh` |
|
||||
| `tests/README.md` | (source-only) Bats test dependency and run instructions |
|
||||
| `tests/validate-adapter.bats` | (source-only) Bats test suite for `scripts/validate-adapter.sh` |
|
||||
|
||||
Rows marked **(source-only)** exist in the authoring source (`.apm/skills/provider-adapter-author/`)
|
||||
but are not present in an installed plugin: `scripts/sync-plugin-content.sh` strips
|
||||
`<category>/<name>/tests` when it generates the flat mirror, because these are dev-time fixtures no
|
||||
plugin host needs to discover (ADR-0017). Run them from a repo checkout, not from an install.
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "git",
|
||||
"version": "1.3.2",
|
||||
"description": "Skills for working with Git \u2014 conventional commits, branch management, pull requests, and feature flow.",
|
||||
"version": "1.3.5",
|
||||
"description": "Skills and agents for working with a local Git clone over the git wire protocol, and for authoring and running the pre-commit hooks that guard it.",
|
||||
"author": {
|
||||
"name": "Defame1297",
|
||||
"email": "defame1297@rkdr.net",
|
||||
|
||||
4
plugins/git/.github/plugin/plugin.json
vendored
4
plugins/git/.github/plugin/plugin.json
vendored
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "git",
|
||||
"version": "1.3.2",
|
||||
"description": "Skills for working with Git \u2014 conventional commits, branch management, pull requests, and feature flow.",
|
||||
"version": "1.3.5",
|
||||
"description": "Skills and agents for working with a local Git clone over the git wire protocol, and for authoring and running the pre-commit hooks that guard it.",
|
||||
"author": {
|
||||
"name": "Defame1297",
|
||||
"email": "defame1297@rkdr.net",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
name: git
|
||||
version: 1.3.2
|
||||
description: Skills for working with Git — conventional commits, branch management, pull requests, and feature flow.
|
||||
version: 1.3.5
|
||||
description: Skills and agents for working with a local Git clone over the git wire protocol, and for authoring and running the pre-commit hooks that guard it.
|
||||
author:
|
||||
name: Defame1297
|
||||
email: defame1297@rkdr.net
|
||||
|
||||
@@ -9,9 +9,10 @@ source_keys:
|
||||
- context7-websites-gitea
|
||||
- context7-gitea-tea-cli
|
||||
|
||||
disallowedTools: Edit, Write, NotebookEdit
|
||||
---
|
||||
|
||||
You are the orchestrator for the gitea plugin — a composable workflow dispatcher designed for other agents to invoke multi-step Gitea operations reliably. Your one job is routing and safety-gating: you do not call `mcp__gitea__*` tools yourself, you delegate to domain skills and enforce confirmation on destructive operations.
|
||||
You are the orchestrator for the gitea plugin — a composable workflow dispatcher designed for other agents to invoke multi-step Gitea operations reliably. Your one job is routing and safety-gating: you do not call `mcp__gitea__*` tools yourself, you delegate to domain skills and enforce confirmation on destructive operations. You never edit files. Every write you cause reaches its target through a domain skill's Gitea API call — never through an edit you make to the local working tree.
|
||||
|
||||
You resolve `owner`/`repo` once per session (via `git remote -v` on `origin`) and carry that forward as session context to every domain skill you dispatch to, rather than making each skill re-resolve it.
|
||||
|
||||
@@ -28,6 +29,7 @@ These are non-negotiable regardless of `confirm` or any skill-local override:
|
||||
- Issues and PRs share one number space. Before dispatching an operation keyed on a bare number, resolve whether it's an issue or a PR yourself (see Number resolution) — never infer the domain from operation phrasing alone.
|
||||
- `list_releases`/`list_tags` default to `per_page: 20` (other domains default to 30) with no server-side auto-pagination — when a caller needs a complete result set, loop `page` upward until a page returns fewer than `per_page` results before returning.
|
||||
- Never commit secrets, credentials, or environment-specific config into any file written via `gitea-files`.
|
||||
- You are read-only against the local working tree. Never create, edit, or delete a local file — not a manifest, not a config, not a scratch note. Local state is the caller's, and you only read it (e.g. `git remote -v`) to resolve context.
|
||||
|
||||
### Number resolution
|
||||
|
||||
@@ -69,7 +71,7 @@ When invoked, you:
|
||||
5. If the operation targets a bare number and the domain isn't specified, run Number resolution above before dispatch
|
||||
6. Invoke the appropriate domain skill via `Skill` with the operation, parameters, and resolved context (`owner`, `repo`)
|
||||
7. Catch and handle Gitea errors: disambiguate 404s (not-found vs. permission-hidden), retry transient failures, loop pagination for `list_releases`/`list_tags` until exhausted
|
||||
8. If recovery succeeds, continue; if not, return error structure with diagnostics and suggestions
|
||||
8. If recovery succeeds, continue; if not, return error structure with diagnostics and suggestions. If the blocker looks trivially fixable by a local edit — a stale `origin` URL, a malformed config, a missing label the repo obviously wants — name that fix in `suggestions` and stop. Do not act on it, and do not route it as a write operation the caller never asked for
|
||||
9. Aggregate all outputs and return as structured JSON
|
||||
|
||||
## Output
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "gitea",
|
||||
"version": "1.3.3",
|
||||
"description": "Skills for managing Gitea repositories \u2014 issues, pull requests, milestones, releases, and wikis.",
|
||||
"version": "1.3.6",
|
||||
"description": "Skills and agents for working with a Gitea forge through its HTTP API \u2014 the forge's own objects, as distinct from the local git clone.",
|
||||
"author": {
|
||||
"name": "Defame1297",
|
||||
"email": "defame1297@rkdr.net",
|
||||
|
||||
4
plugins/gitea/.github/plugin/plugin.json
vendored
4
plugins/gitea/.github/plugin/plugin.json
vendored
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "gitea",
|
||||
"version": "1.3.3",
|
||||
"description": "Skills for managing Gitea repositories \u2014 issues, pull requests, milestones, releases, and wikis.",
|
||||
"version": "1.3.6",
|
||||
"description": "Skills and agents for working with a Gitea forge through its HTTP API \u2014 the forge's own objects, as distinct from the local git clone.",
|
||||
"author": {
|
||||
"name": "Defame1297",
|
||||
"email": "defame1297@rkdr.net",
|
||||
|
||||
@@ -9,9 +9,10 @@ source_keys:
|
||||
- context7-websites-gitea
|
||||
- context7-gitea-tea-cli
|
||||
|
||||
disallowedTools: Edit, Write, NotebookEdit
|
||||
---
|
||||
|
||||
You are the orchestrator for the gitea plugin — a composable workflow dispatcher designed for other agents to invoke multi-step Gitea operations reliably. Your one job is routing and safety-gating: you do not call `mcp__gitea__*` tools yourself, you delegate to domain skills and enforce confirmation on destructive operations.
|
||||
You are the orchestrator for the gitea plugin — a composable workflow dispatcher designed for other agents to invoke multi-step Gitea operations reliably. Your one job is routing and safety-gating: you do not call `mcp__gitea__*` tools yourself, you delegate to domain skills and enforce confirmation on destructive operations. You never edit files. Every write you cause reaches its target through a domain skill's Gitea API call — never through an edit you make to the local working tree.
|
||||
|
||||
You resolve `owner`/`repo` once per session (via `git remote -v` on `origin`) and carry that forward as session context to every domain skill you dispatch to, rather than making each skill re-resolve it.
|
||||
|
||||
@@ -28,6 +29,7 @@ These are non-negotiable regardless of `confirm` or any skill-local override:
|
||||
- Issues and PRs share one number space. Before dispatching an operation keyed on a bare number, resolve whether it's an issue or a PR yourself (see Number resolution) — never infer the domain from operation phrasing alone.
|
||||
- `list_releases`/`list_tags` default to `per_page: 20` (other domains default to 30) with no server-side auto-pagination — when a caller needs a complete result set, loop `page` upward until a page returns fewer than `per_page` results before returning.
|
||||
- Never commit secrets, credentials, or environment-specific config into any file written via `gitea-files`.
|
||||
- You are read-only against the local working tree. Never create, edit, or delete a local file — not a manifest, not a config, not a scratch note. Local state is the caller's, and you only read it (e.g. `git remote -v`) to resolve context.
|
||||
|
||||
### Number resolution
|
||||
|
||||
@@ -69,7 +71,7 @@ When invoked, you:
|
||||
5. If the operation targets a bare number and the domain isn't specified, run Number resolution above before dispatch
|
||||
6. Invoke the appropriate domain skill via `Skill` with the operation, parameters, and resolved context (`owner`, `repo`)
|
||||
7. Catch and handle Gitea errors: disambiguate 404s (not-found vs. permission-hidden), retry transient failures, loop pagination for `list_releases`/`list_tags` until exhausted
|
||||
8. If recovery succeeds, continue; if not, return error structure with diagnostics and suggestions
|
||||
8. If recovery succeeds, continue; if not, return error structure with diagnostics and suggestions. If the blocker looks trivially fixable by a local edit — a stale `origin` URL, a malformed config, a missing label the repo obviously wants — name that fix in `suggestions` and stop. Do not act on it, and do not route it as a write operation the caller never asked for
|
||||
9. Aggregate all outputs and return as structured JSON
|
||||
|
||||
## Output
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
name: gitea
|
||||
version: 1.3.3
|
||||
description: Skills for managing Gitea repositories — issues, pull requests, milestones, releases, and wikis.
|
||||
version: 1.3.6
|
||||
description: Skills and agents for working with a Gitea forge through its HTTP API — the forge's own objects, as distinct from the local git clone.
|
||||
author:
|
||||
name: Defame1297
|
||||
email: defame1297@rkdr.net
|
||||
|
||||
@@ -5,9 +5,11 @@ description: Orchestrates apm package/marketplace operations for other agents. I
|
||||
|
||||
source_keys:
|
||||
- context7-microsoft-apm
|
||||
|
||||
disallowedTools: Edit, Write, NotebookEdit
|
||||
---
|
||||
|
||||
You are the orchestrator for apm package/marketplace operations — a composable workflow dispatcher designed for other agents to invoke multi-step `apm` operations reliably, especially the same operation repeated across several packages in a monorepo-hybrid layout. Your one job is routing and safety-gating: you do not decide manifest content yourself, you delegate to `apm-workflow` and enforce confirmation on irreversible operations.
|
||||
You are the orchestrator for apm package/marketplace operations — a composable workflow dispatcher designed for other agents to invoke multi-step `apm` operations reliably, especially the same operation repeated across several packages in a monorepo-hybrid layout. Your one job is routing and safety-gating: you do not decide manifest content yourself, you delegate to `apm-workflow` and enforce confirmation on irreversible operations. You never edit files. Every manifest or primitive that changes under your dispatch is written by `apm-workflow` or by `apm` itself — never by an edit you make.
|
||||
|
||||
You resolve the package root once per dispatched operation (the directory containing that package's `apm.yml`) and carry it forward as session context rather than making every call re-resolve it.
|
||||
|
||||
@@ -21,6 +23,7 @@ These are non-negotiable regardless of `confirm` or any skill-local override:
|
||||
- `apm.yml`'s `type:` field constrains what `.apm/` may contain — when scaffolding (`init-package`), set `type:` before any primitive content is added; do not defer it.
|
||||
- A clean plain `apm audit` is not a CI-equivalent pass — if the caller's intent is a CI gate, dispatch `audit-ci`, not `audit`.
|
||||
- Check the `apm experimental enable registries` precondition before dispatching any operation that depends on a named registry, and fail with a clear diagnostic rather than silently no-op'ing like apm itself does — see apm-workflow/SKILL.md Gotchas for the underlying constraint.
|
||||
- You are read-only against the working tree. Never create, edit, or delete a file — not an `apm.yml`, not a `.apm/` primitive, not compiled output, not a scratch note. `edit-config` is an operation you *route* to `apm-workflow`, never one you perform: dispatching it is allowed only when the caller asked for that edit, never as your own repair of something you noticed.
|
||||
|
||||
When invoked, you:
|
||||
1. Parse the incoming workflow request (operation type, parameters, target package(s), context overrides)
|
||||
@@ -52,7 +55,7 @@ When invoked, you:
|
||||
4. Verify `apm --version` succeeds; if not, fail with a diagnostic pointing to `apm-install`
|
||||
5. Invoke `apm-workflow` via `Skill` with the resolved action, `package_root`, and parameters
|
||||
6. If fanning across multiple packages, dispatch independent packages in parallel when no shared state or ordering dependency exists between them; loop package-by-package (strictly sequential) only for packages with a real dependency on another package's completion. Either way, collect per-package results and failures rather than aborting on the first failure
|
||||
7. Catch and handle apm errors: retry once for a dependency-not-yet-scaffolded failure after the caller confirms the dependency exists; otherwise return error structure with diagnostics
|
||||
7. Catch and handle apm errors: retry once for a dependency-not-yet-scaffolded failure after the caller confirms the dependency exists; otherwise return error structure with diagnostics. If the failure looks trivially fixable by a one-line manifest edit — a missing `category:`, a typo'd `source:`, a version that disagrees between a package and the catalog — name that fix in `suggestions` and stop. Do not apply it yourself and do not self-dispatch an `edit-config` to apply it
|
||||
8. Aggregate all outputs and return as structured JSON
|
||||
|
||||
## Output
|
||||
|
||||
60
plugins/kyberforge/.apm/hooks/check-apm-current.sh
Executable file
60
plugins/kyberforge/.apm/hooks/check-apm-current.sh
Executable file
@@ -0,0 +1,60 @@
|
||||
#!/usr/bin/env bash
|
||||
# SessionStart: keep an apm-consumed install level with its remote.
|
||||
#
|
||||
# Packages declared as unpinned git refs resolve against the remote default
|
||||
# branch, so the deployed .claude/skills/ and .claude/agents/ go stale the
|
||||
# moment anyone merges. The staleness bites when a session loads skills, which
|
||||
# is why this runs at SessionStart rather than off a git hook — a pull is
|
||||
# neither necessary nor sufficient for the install to have drifted.
|
||||
#
|
||||
# Refreshes in place and asks the host to re-scan, so the running session picks
|
||||
# the new content up without a restart.
|
||||
#
|
||||
# Inert in any project that does not consume packages through apm.
|
||||
set -uo pipefail
|
||||
|
||||
# Anchor on the project root, not the session's cwd. Claude Code exports
|
||||
# CLAUDE_PROJECT_DIR for SessionStart hooks; a session opened in a subdirectory
|
||||
# would otherwise miss the lockfile, no-op silently, and — worse — run the apm
|
||||
# calls below against that wrong directory. Fall back to the cwd when the
|
||||
# variable is absent, which keeps the hook inert-but-harmless under a host that
|
||||
# does not set it.
|
||||
project_dir="${CLAUDE_PROJECT_DIR:-$PWD}"
|
||||
|
||||
# No lockfile means nothing was installed through apm here — e.g. a host that
|
||||
# installed this plugin natively. Say nothing and cost nothing.
|
||||
[[ -f "$project_dir/apm.lock.yaml" ]] || exit 0
|
||||
command -v apm > /dev/null 2>&1 || exit 0
|
||||
|
||||
# Every apm call below must see the same directory the guard just checked —
|
||||
# `apm outdated` and `apm update` both resolve the lockfile from the cwd.
|
||||
cd "$project_dir" || exit 0
|
||||
|
||||
# `apm outdated` exits 0 whether or not anything is stale, so the answer has to
|
||||
# come from its output. ~0.7s against six remote refs; a hung remote must not
|
||||
# hold the session open.
|
||||
#
|
||||
# There is no --json/machine-readable flag on `apm outdated` (verified against
|
||||
# apm 0.28.0), so the phrase match is forced rather than chosen. Note the
|
||||
# singular: apm prints "1 outdated dependency found" when exactly one package is
|
||||
# behind, so matching only "dependencies" would silently miss a one-package
|
||||
# drift. tests/test-apm-current-hook.sh pins both spellings against the real apm.
|
||||
outdated_output="$(timeout 60 apm outdated 2>&1)" || exit 0
|
||||
grep -qE 'outdated dependenc(y|ies) found' <<< "$outdated_output" || exit 0
|
||||
|
||||
stale_count="$(grep -oE '[0-9]+ outdated dependenc(y|ies) found' <<< "$outdated_output" | grep -oE '^[0-9]+' || true)"
|
||||
[[ "$stale_count" =~ ^[0-9]+$ ]] || stale_count="some"
|
||||
|
||||
# Only ever emit fixed text plus a digit-checked count — never interpolate
|
||||
# command output into the JSON, which would need escaping this cannot do safely.
|
||||
emit() {
|
||||
printf '{"hookSpecificOutput":{"hookEventName":"SessionStart","reloadSkills":%s,"additionalContext":"%s"}}\n' "$1" "$2"
|
||||
}
|
||||
|
||||
if timeout 300 apm update --yes > /dev/null 2>&1; then
|
||||
emit true "apm install was ${stale_count} package(s) behind the remote default branch and has been refreshed automatically; skills and agents were redeployed and re-scanned. apm.lock.yaml has been rewritten and is now a modified file in the working tree - commit it or discard it deliberately."
|
||||
else
|
||||
emit false "apm install is ${stale_count} package(s) behind the remote default branch and the automatic refresh failed. Deployed skills and agents may be stale. Run: apm update --yes"
|
||||
fi
|
||||
|
||||
exit 0
|
||||
@@ -1,3 +1,16 @@
|
||||
{
|
||||
"hooks": {}
|
||||
"hooks": {
|
||||
"SessionStart": [
|
||||
{
|
||||
"hooks": [
|
||||
{
|
||||
"command": "${CLAUDE_PLUGIN_ROOT}/.apm/hooks/check-apm-current.sh",
|
||||
"timeout": 380,
|
||||
"type": "command"
|
||||
}
|
||||
],
|
||||
"matcher": "startup"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,29 +1,40 @@
|
||||
# agent-audit
|
||||
|
||||
Audits an agent definition for correctness and quality — a single vendor-neutral file at
|
||||
Audits an agent definition for correctness and quality against the Claude Code and Copilot agent
|
||||
references and the house context-budget contract (ADR-0020) — a single vendor-neutral file at
|
||||
plugin/APM scope, or a Claude Code and Copilot file pair at project/user scope.
|
||||
|
||||
## What it does
|
||||
|
||||
At **plugin/APM scope**, accepts the single `.apm/agents/<name>.agent.md` file — there is no
|
||||
counterpart. Structural checks via `validate.sh` hard-`FAIL` any frontmatter field outside the
|
||||
vendor-neutral allowlist (`name`, `description`, `model`, `source_keys` — the last for
|
||||
provenance tracking, checked separately by `validate-provenance.sh` against `sources.md`; see
|
||||
ADR-0016), since `apm compile`
|
||||
copies frontmatter verbatim to both harnesses and an unsafe field can't be silently dropped for
|
||||
just one of them.
|
||||
1. Runs `scripts/validate.sh` and `scripts/validate-provenance.sh` for structural and provenance
|
||||
checks, plus `scripts/vale-wrap.sh` — a Vale prefilter that deterministically flags
|
||||
non-imperative description openers, composition and architecture notes, vague wording, padding
|
||||
phrases, "There is/are" sentence openers, and CC-specific "Use proactively" phrasing in a
|
||||
Copilot or vendor-neutral description
|
||||
2. Reads the agent file, and its counterpart when one exists, then loads the contract for its scope
|
||||
3. Applies qualitative checks across description, body, delegation and comment discipline, loading
|
||||
one rubric from `references/` per group
|
||||
4. Outputs a compact findings report — findings only, grouped by dimension, each with Why and Fix —
|
||||
and a result block with handoff to `agent-author`
|
||||
|
||||
At **project/user scope**, accepts either file in a CC `.md` / Copilot `.agent.md` pair, derives
|
||||
the counterpart automatically, and validates both. Runs structural checks via `validate.sh`
|
||||
(required fields, kebab-case name, no placeholders, no CC-only fields in the Copilot file, no
|
||||
Copilot-only fields in the CC file), provenance chain validation via `validate-provenance.sh`
|
||||
(checks `source_keys` against `sources.md` at the plugin root — plugin/APM scope only), then
|
||||
qualitative checks on description phrasing and system prompt quality. Step 1 also runs a
|
||||
Vale-based prose sub-check via `vale-wrap.sh` against both files of the pair, using the
|
||||
`Kyberforge` style (both files) and `KyberforgeCopilot` style (Copilot file only) — every alert
|
||||
is a `FAIL`, cited by rule ID — falling back to Step 2 judgment when the `vale` binary is
|
||||
unavailable or reports `0 files` scanned. Produces a compact findings report in the same format
|
||||
as `skill-audit`.
|
||||
Two things follow from ADR-0020 and are easy to get backwards. Agents take the **same** description
|
||||
gates a skill takes — 250 characters SUGGESTION, 400 FAIL, since a `name` + `description` is
|
||||
preloaded into every session either way — and **no body word gate at all**, because an agent body
|
||||
becomes the system prompt of a fresh context rather than competing with the caller's live
|
||||
conversation. Body length is judged through the delegation check instead: an agent body that
|
||||
restates a procedure owned by a skill it can invoke is a FAIL, because a plugin-scope agent has no
|
||||
sibling `references/` directory to disclose to and can only delegate.
|
||||
|
||||
At **plugin/APM scope** the audit accepts the single `.apm/agents/<name>.agent.md` file — there is
|
||||
no counterpart, and pair consistency does not apply. `validate.sh` hard-`FAIL`s any frontmatter
|
||||
field outside the vendor-neutral allowlist, since `apm compile` copies frontmatter verbatim to both
|
||||
harnesses and an unsafe field cannot be silently dropped for just one of them. The allowlist lives
|
||||
in the `apm-agent-allowlist` section of `references/field-inventory.md`, is read from there as data
|
||||
by the script, and is deliberately not restated anywhere else in this skill (ADR-0009).
|
||||
|
||||
At **project/user scope** the audit accepts either file in a CC `.md` / Copilot `.agent.md` pair,
|
||||
derives the counterpart automatically, and validates both, including the field-leakage checks in
|
||||
each direction.
|
||||
|
||||
## Usage
|
||||
|
||||
@@ -39,19 +50,29 @@ Pass the path to either agent file as the argument.
|
||||
|------|---------|
|
||||
| `SKILL.md` | Skill instructions for agents |
|
||||
| `assets/vale/.vale.ini` | Vale config: scopes `Kyberforge` to `**/agents/*.md`, `Kyberforge`+`KyberforgeCopilot` to `**/*.agent.md` |
|
||||
| `assets/vale/styles/Kyberforge/DescriptionOpener.yml` | Flags descriptions opening with "This skill/agent" instead of an imperative "Use when..." |
|
||||
| `assets/vale/styles/Kyberforge/CompositionNote.yml` | Flags composition and architecture notes in a description ("cross-cutting", "entry point", "composes", "rather than duplicating") that belong in README.md |
|
||||
| `assets/vale/styles/Kyberforge/DescriptionOpener.yml` | Flags descriptions opening with "This..." instead of an imperative "Use when..." |
|
||||
| `assets/vale/styles/Kyberforge/PaddingPhrase.yml` | Flags generic "see references/ for info" pointers instead of specific file references |
|
||||
| `assets/vale/styles/Kyberforge/SentenceOpenerThereIs.yml` | Flags sentences opening with "There is/are" instead of naming the subject directly |
|
||||
| `assets/vale/styles/Kyberforge/VagueWording.yml` | Flags vague capability wording ("helps with", "utilize", "assists with", "used for") in descriptions |
|
||||
| `assets/vale/styles/KyberforgeCopilot/ProactivePhrase.yml` | Flags CC-specific "Use proactively" phrasing with no effect in Copilot descriptions |
|
||||
| `references/README.md` | Directory documentation for references/ |
|
||||
| `references/description-quality.md` | Qualitative guide for borderline description findings |
|
||||
| `references/field-inventory.md` | Authoritative list of valid CC and Copilot agent fields |
|
||||
| `references/description-quality.md` | Rubric for the description dimension — three-part shape, the 250/400-character budget, the hand-invoked contract, and the internal-mechanics FAIL |
|
||||
| `references/body-and-delegation.md` | Rubric for the body, delegation and comment-discipline dimensions — the delegation FAIL and why agents take no body word gate |
|
||||
| `references/scope-plugin-apm.md` | Scope contract for a single vendor-neutral APM agent file — allowlist, dimension routing, and the dimensions that do not apply |
|
||||
| `references/scope-project-user.md` | Scope contract for a CC / Copilot pair — counterpart derivation, provider field rules, pair consistency |
|
||||
| `references/validation-scripts.md` | Loaded only when a Step 1 script fails or cannot run — scope-detection walk-up, manual fallback checks, known script failures |
|
||||
| `references/field-inventory.md` | Authoritative field lists read as data by `validate.sh`: valid CC and Copilot agent fields, and the vendor-neutral plugin/APM-scope allowlist |
|
||||
| `references/sources.md` | Research provenance for skill content |
|
||||
| `scripts/README.md` | Directory documentation for scripts/ |
|
||||
| `scripts/validate.sh` | Structural validation script for agent file pairs |
|
||||
| `scripts/validate-provenance.sh` | Provenance chain validation script for agent pairs against `sources.md` (plugin root) |
|
||||
| `scripts/validate.sh` | Structural validator — required fields, name format, placeholder detection, the ADR-0020 description budget, and the field rules for the detected scope |
|
||||
| `scripts/validate-provenance.sh` | Provenance chain validation against `sources.md` at the package root (plugin/APM scope only) |
|
||||
| `scripts/vale-wrap.sh` | Drop-in `vale` wrapper that works around a frontmatter-description NLP scope limitation |
|
||||
| `tests/README.md` | Bats test dependency and run instructions |
|
||||
| `tests/validate.bats` | Bats tests for validate.sh |
|
||||
| `tests/validate-provenance.bats` | Bats tests for validate-provenance.sh |
|
||||
| `tests/README.md` | (source-only) Bats test dependency and run instructions |
|
||||
| `tests/validate.bats` | (source-only) Bats tests for validate.sh |
|
||||
| `tests/validate-provenance.bats` | (source-only) Bats tests for validate-provenance.sh |
|
||||
|
||||
Rows marked **(source-only)** exist in the authoring source (`.apm/skills/agent-audit/`) but are
|
||||
not present in an installed plugin: `scripts/sync-plugin-content.sh` strips
|
||||
`<category>/<name>/tests` when it generates the flat mirror, because these are dev-time fixtures no
|
||||
plugin host needs to discover (ADR-0017). Run them from a repo checkout, not from an install.
|
||||
|
||||
@@ -1,18 +1,10 @@
|
||||
---
|
||||
name: agent-audit
|
||||
description: >
|
||||
Use when the user wants to review an agent definition they wrote, says "audit this
|
||||
agent", "check if my agent follows best practices", "review my agent file", or wants
|
||||
to know if an agent pair is ready to ship — even if they don't use the word "audit".
|
||||
Also invoke proactively after directly hand-editing an agent file pair outside
|
||||
agent-author — an unaudited hand-edit is the same risk as unreviewed code.
|
||||
Audits a Claude Code .md and Copilot .agent.md agent file pair across six dimensions:
|
||||
structural validation, provider safety, description quality, body quality, comment
|
||||
discipline, and pair consistency — plus provenance chain validation. Produces a
|
||||
compact findings report
|
||||
(findings only, no PASS noise) with Why and Fix per finding. Do not use to fix agent
|
||||
files — use /agent-author instead. Do not use to audit SKILL.md files — use
|
||||
/skill-audit instead.
|
||||
Use when the user wants an agent definition audited — "audit this agent",
|
||||
"review my agent file", "is this ready to ship" — or after hand-editing an
|
||||
agent outside agent-author. Not applying fixes -> agent-author. Not a skill
|
||||
directory -> skill-audit.
|
||||
allowed-tools: Bash Read
|
||||
metadata:
|
||||
category: factory
|
||||
@@ -26,80 +18,67 @@ metadata:
|
||||
|
||||
## Gotchas
|
||||
|
||||
- The unit of authoring at project/user scope is always a pair (CC `.md` + Copilot `.agent.md`). A missing counterpart is a FAIL under the kyberforge project convention at those scopes — neither the CC nor the Copilot platform itself requires a counterpart file. Label such findings as project convention violations, not platform spec failures. **At plugin/APM scope there is no pair** — the unit of authoring is a single vendor-neutral `.apm/agents/<name>.agent.md` file, and Pair Consistency does not apply there at all (see below).
|
||||
- Scope is detected by walking up from the agent file's directory: at each level, if `apm.yml` exists AND contains a top-level `type: instructions|skill|hybrid|prompts` line, that directory is an APM package root — plugin/APM scope. A `type:`-less `apm.yml` is marketplace-only (see `docs/research/docs/microsoft-apm/monorepo-and-repo-shapes.md`) — skip it and keep walking up. Otherwise, if `.git` is a directory at that level, stop there — project scope. If neither is found before the filesystem root, fall back to user scope at `$HOME`. `plugin.json`/`.claude-plugin/plugin.json` are no longer scope signals for this skill — a directory with only a `plugin.json` and no `apm.yml` falls through to project (or user) scope.
|
||||
- `references/field-inventory.md` must exist for `validate.sh` to run. The script exits with an error if it is missing.
|
||||
- Do not output findings while auditing — gather internally, surface in Step 3 report.
|
||||
- Do not narrate PASS/FAIL per check while auditing. Gather findings internally and surface them only in the Step 4 report. Narrating each check as you go is the default failure mode here.
|
||||
- Agents take the same 250/400-character description gates as skills and **no body word gate at all** — an agent body becomes the system prompt of a fresh context, so the 900-word skill ceiling does not transfer. Judge an over-long agent body through the delegation check, never by word count.
|
||||
- At plugin/APM scope the agent is a single vendor-neutral file by design: never raise a pair-consistency finding there, and provider safety stops meaning Claude-Code-versus-Copilot field leakage.
|
||||
- Vale reporting `0 files` scanned means NOT RUN, not clean. Fall back to full Step 3 judgment for every dimension it would have covered.
|
||||
|
||||
## Step 1 — Run structural validation
|
||||
## Step 1 — Deterministic checks
|
||||
|
||||
Resolve all three paths against this skill's own directory so they work from a repo checkout and an installed plugin cache alike. Run exactly:
|
||||
|
||||
```bash
|
||||
bash scripts/validate.sh <path-to-agent-file>
|
||||
bash scripts/validate-provenance.sh <path-to-agent-file>
|
||||
scripts/vale-wrap.sh <path-to-cc-file> <path-to-copilot-file> # project/user scope
|
||||
scripts/vale-wrap.sh <path-to-apm-agent-file> # plugin/APM scope — single file
|
||||
bash scripts/validate.sh <agent-file>
|
||||
bash scripts/validate-provenance.sh <agent-file>
|
||||
scripts/vale-wrap.sh <agent-file> [<counterpart-file>]
|
||||
```
|
||||
|
||||
The script accepts either the CC file, the Copilot file, or (at plugin/APM scope) the single `.apm/agents/<name>.agent.md` file. It detects provider from extension and scope from the walk-up above, then runs the checks for that scope.
|
||||
`validate.sh` takes either half of a project/user-scope pair or the single plugin/APM-scope file, detects the provider from the extension and the scope by walking up, then checks required fields, kebab-case `name`, `FILL IN:` placeholders, template HTML comments left in frontmatter, the ADR-0020 description budget (250 chars SUGGESTION, 400 FAIL, measured on the folded YAML value) and the fields that scope permits. Its findings become the `### Structure` dimension — its FAILs and its SUGGESTIONs both — except the ones the Step 2 scope contract re-routes.
|
||||
|
||||
At **project/user scope** it derives the counterpart and runs the existing pair-based checks. Note FAILs and SUGGESTIONs for the `### Structure` and `### Provider safety` report dimensions. Findings about missing fields, bad name format, empty body, or missing frontmatter → `### Structure`. Findings about CC-only fields in a Copilot file, Copilot-only fields in a CC file, body length, or subagent-unavailable tools → `### Provider safety`. A missing counterpart file → `### Pair consistency`.
|
||||
If a validation script fails or cannot run — Bash denied, `python3` or `vale` absent, `references/field-inventory.md` missing — read `references/validation-scripts.md`; what these scripts measure is not reproducible by reading.
|
||||
|
||||
At **plugin/APM scope** there is no counterpart — the script instead checks the single file's frontmatter against the `apm-agent-allowlist` in `references/field-inventory.md` (`name`, `description`, `model`, `source_keys` — nothing else; `source_keys` is provenance metadata, not a provider-specific field, and is validated separately by `validate-provenance.sh` against `sources.md`). Findings about missing fields, bad name format, name/filename-stem mismatch, empty body, or missing frontmatter → `### Structure`, same as project/user scope. Findings about any field outside the allowlist (e.g. `tools`, or any Claude-only/Copilot-only field carried over from a hand-edit) and body length → `### Provider safety` — but the dimension's meaning shifts here: it is no longer a CC-vs-Copilot field-leakage check, it's a vendor-neutral-field-allowlist check, since `apm compile` verbatim-copies this file's frontmatter to every target and there is no per-target integrator to reconcile a CC-only or Copilot-only field (ADR-0016). `### Pair consistency` never applies at this scope — the script never emits a missing-counterpart FAIL here, because there is nothing to pair by design.
|
||||
`validate-provenance.sh` prints nothing on success and runs at plugin/APM scope only, exiting 0 silently elsewhere. Its FAIL findings become a separate `### Provenance` dimension, and it emits Why and Fix itself — surface those verbatim.
|
||||
|
||||
`vale-wrap.sh` ships inside this skill's own `scripts/` — resolve it relative to this skill's directory the same way `scripts/validate.sh` is resolved above, so the invocation works whether this skill is running from this repo or from an installed plugin cache. Pass no `--config`: handed none, the wrapper loads its own sibling `assets/vale/.vale.ini`, located from the script's path rather than from the cwd. Adding an explicit relative `--config` breaks exactly the case the self-location covers — a resolved script path plus an unresolved config path yields `E100 Runtime error ... does not exist`, exit 2, which the fallback below then misreads as "vale unavailable". At project/user scope, run it against both files of the pair (not just the one passed in); at plugin/APM scope, run it against the single file. `Kyberforge` applies to all of these files via the `**/agents/*.md` glob; `KyberforgeCopilot` applies to any `*.agent.md` file — including the plugin/APM-scope file, which already has that extension — via the `**/*.agent.md` glob, since its one rule (`Use proactively`) flags CC-specific phrasing that's meaningless in a vendor-neutral or Copilot description. Every Vale alert is a `FAIL` — all rules are graded `error` — so report each one in the `### Description` / `### Body` dimensions citing its rule ID (e.g. `KyberforgeCopilot.ProactivePhrase`). Skip and fall back to Step 2 judgment if the `vale` binary is unavailable. If Vale reports `0 files` scanned, treat the pass as NOT RUN — not as clean — and fall back to full Step 2 judgment for the dimensions it would have covered.
|
||||
`vale-wrap.sh` applies the bundled `Kyberforge` style as a prefilter. Pass no `--config`; the wrapper locates its own. At project/user scope pass both files of the pair, not only the one you were handed. Every rule is graded `error`, so every alert is a FAIL. Report each one citing its rule ID, filed under the dimension it belongs to, and do not re-derive it by judgment:
|
||||
|
||||
`validate-provenance.sh` operates at plugin/APM scope only — it walks up from the agent file's directory the same way `validate.sh` does (nearest ancestor `apm.yml` with a top-level `type:` field; skip a `type:`-less marketplace-only `apm.yml`; stop at `.git` or the filesystem root) and exits 0 silently if that walk doesn't land on a package root, or when no provenance data exists. When it does apply, it validates the chain between the single file's own `source_keys` and the package-scoped `sources.md` (package root — see ADR-0010). Note FAILs from this script for the `### Provenance` dimension — surface them verbatim with Why and Fix.
|
||||
| Rule | Dimension |
|
||||
|---|---|
|
||||
| `Kyberforge.DescriptionOpener`, `Kyberforge.CompositionNote`, `Kyberforge.VagueWording`, `KyberforgeCopilot.ProactivePhrase` | description |
|
||||
| `Kyberforge.SentenceOpenerThereIs`, `Kyberforge.PaddingPhrase` | body |
|
||||
|
||||
If the scripts cannot run (Bash denied, python3 unavailable), perform checks manually. At project/user scope: counterpart file exists, required fields present (`name`, `description`, non-empty body), `name` is kebab-case, Copilot CLI `.agent.md` `name` must match filename stem (CC files are exempt — the CC platform does not require name to match filename), no `FILL IN:` placeholders, no CC-only fields in Copilot file, no Copilot-only fields in CC file (read `references/field-inventory.md` for the authoritative field lists). At plugin/APM scope: required fields present (`name`, `description`, non-empty body), `name` is kebab-case and matches the filename stem, no `FILL IN:` placeholders, no frontmatter field outside `name`/`description`/`model`/`source_keys` (read the `apm-agent-allowlist` section of `references/field-inventory.md`; `source_keys` carries provenance metadata, checked separately by `validate-provenance.sh` against `sources.md`).
|
||||
## Step 2 — Read the agent and load its scope contract
|
||||
|
||||
## Step 2 — Qualitative checks
|
||||
Read the agent file end to end, and at project/user scope its counterpart too. A path containing `.apm/agents/` is plugin/APM scope; anything else is project or user scope. Each contract names the dimensions that apply there and where `validate.sh` findings other than Structure belong:
|
||||
|
||||
Read both agent files. Work through each dimension internally. Collect findings only; report in Step 3.
|
||||
| Scope | Read |
|
||||
|---|---|
|
||||
| plugin/APM | `references/scope-plugin-apm.md` |
|
||||
| project, user | `references/scope-project-user.md` |
|
||||
|
||||
**Description (both files):**
|
||||
- Action-verb opening: description starts with a verb ("Reviews...", "Analyzes...", "Generates...") — FAIL if absent. Vale's `Kyberforge.DescriptionOpener` alert flags the specific known-bad "This agent..." opener directly; verifying an arbitrary opening word is genuinely a strong verb still requires judgment.
|
||||
- Specificity: is the trigger condition stated precisely? — SUGGESTION if vague. Vale's `Kyberforge.VagueWording` alert covers known filler ("helps with", "utilize", ...) directly; report those as FAILs without re-deriving by judgment.
|
||||
- `Use proactively` in a Copilot description: Vale's `KyberforgeCopilot.ProactivePhrase` alert (Copilot file only) flags this directly — report it without re-deriving by judgment.
|
||||
## Step 3 — Qualitative audit
|
||||
|
||||
If a description finding is borderline, read `references/description-quality.md`.
|
||||
Load a dimension's rubric before judging that dimension.
|
||||
|
||||
**Body:**
|
||||
- Direct role instruction: system prompt opens with `You are a [role]. When invoked, [action].` — SUGGESTION if absent
|
||||
- One job per agent: system prompt describes a single bounded task — SUGGESTION if scope appears unbounded
|
||||
- Generic, non-specific reference pointers to the `references/` directory: Vale's `Kyberforge.PaddingPhrase` alert flags this directly — report it without re-deriving by judgment
|
||||
- Sentences that open with "There is"/"There are": Vale's `Kyberforge.SentenceOpenerThereIs` alert flags this directly — report it without re-deriving by judgment
|
||||
| Dimension | Read |
|
||||
|---|---|
|
||||
| description | `references/description-quality.md` |
|
||||
| body, delegation, comment-discipline | `references/body-and-delegation.md` |
|
||||
|
||||
**Body/Frontmatter comments:**
|
||||
- Inspect each comment block in the YAML frontmatter. For each comment, apply: *"Would the agent get this wrong without this comment?"* Flag any that answer "no" as padding.
|
||||
- Look for patterns like `# Optional. <long explanation>` or extensive inline guidance (more than 1–2 lines per field) that should be condensed or removed before shipping.
|
||||
- This mirrors skill-audit's body-discipline check but applies to template documentation in the frontmatter — template guidance belongs in development; agent-ready files should have minimal comments.
|
||||
Cite file and line number for every finding.
|
||||
|
||||
**Pair consistency (cross-file) — project/user scope only:**
|
||||
- Both files exist — FAIL if counterpart is missing (kyberforge project convention; not a platform requirement from either CC or Copilot — label as such)
|
||||
- The following checks are covered automatically by `validate.sh`; apply them manually only when the script cannot run: both system prompt bodies non-empty — FAIL if either is empty
|
||||
- **Does not apply at plugin/APM scope** — there is only one file, by design; do not raise a Pair Consistency finding there under any circumstance.
|
||||
## Step 4 — Report
|
||||
|
||||
**Unexpressable Claude-only behavior — plugin/APM scope only:**
|
||||
- Read the description and body. If either implies a need the vendor-neutral frontmatter can no longer express — tool restriction, `isolation`, `memory`, or another Claude-only behavior that a hand-authored CC file could have declared — flag it as a SUGGESTION, never a FAIL. This is a known upstream schema limitation (APM's agent primitive has no per-target compile integrator, so `tools:`/`isolation`/etc. can't be emitted safely to both CC and Copilot — ADR-0016), not an authoring mistake. The finding exists to give the author visibility into the gap, not to imply the schema can be made to do something it can't.
|
||||
- Example: a body that says "only use Read and Grep, never Edit" but the frontmatter has no `tools` field to enforce it — SUGGESTION, not FAIL.
|
||||
|
||||
## Step 3 — Report
|
||||
|
||||
Open with a coverage line. At project/user scope:
|
||||
Open with a coverage line naming every dimension checked. At project/user scope:
|
||||
|
||||
```text
|
||||
Checked: structure · provider-safety · description · body · comment-discipline · pair-consistency · provenance
|
||||
Checked: structure · provider-safety · description · body · delegation · comment-discipline · pair-consistency · provenance
|
||||
```
|
||||
|
||||
At plugin/APM scope, omit `pair-consistency` — it does not apply when there is no pair:
|
||||
At plugin/APM scope, drop `pair-consistency` — there is no pair to check.
|
||||
|
||||
```text
|
||||
Checked: structure · provider-safety · description · body · comment-discipline · provenance
|
||||
```
|
||||
Then output only the dimensions that have findings, grouped under H3 headings, FAILs before SUGGESTIONs within each. Omit clean dimensions — their absence is what confirms they passed.
|
||||
|
||||
Then output only dimensions that have findings, grouped under H3 headings, FAILs before SUGGESTIONs within each dimension. Omit clean dimensions entirely. `### Provenance` findings are sourced verbatim from `validate-provenance.sh` output — copy them without rephrasing.
|
||||
|
||||
For each finding:
|
||||
Each finding:
|
||||
|
||||
```text
|
||||
FAIL/SUGGESTION <finding> — file:line
|
||||
@@ -107,18 +86,4 @@ FAIL/SUGGESTION <finding> — file:line
|
||||
Fix: <exact change — quote before/after where applicable>
|
||||
```
|
||||
|
||||
Close with:
|
||||
|
||||
```text
|
||||
## Result
|
||||
|
||||
PASS
|
||||
PASS · P info
|
||||
PASS (N suggestions)
|
||||
PASS (N suggestions) · P info
|
||||
FAIL (N fails · M suggestions)
|
||||
FAIL (N fails · M suggestions) · P info
|
||||
Run /agent-author to address findings.
|
||||
```
|
||||
|
||||
Omit `Run /agent-author to address findings.` when there are no findings at all. Do not apply fixes — report and propose only.
|
||||
Close with a `## Result` block holding one line: `PASS`, `PASS (N suggestions)`, or `FAIL (N fails · M suggestions)`, each optionally followed by ` · P info`. INFO findings are observational and never change PASS/FAIL; omit `· P info` when there are none. Add a second line, `Run agent-author to address findings.`, whenever there is at least one finding. Do not apply fixes — report and propose only.
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
extends: existence
|
||||
message: "Composition or architecture note in a description: '%s' — a description carries a trigger, one capability clause and a boundary clause only; move this to README.md"
|
||||
level: error
|
||||
scope: text.frontmatter.description
|
||||
ignorecase: true
|
||||
tokens:
|
||||
- cross-cutting
|
||||
- shared (skill|agent)
|
||||
- human-facing
|
||||
- entry[- ]point
|
||||
- composes
|
||||
- rather than duplicating
|
||||
- replaces the (old|former|previous)
|
||||
@@ -4,4 +4,4 @@ level: error
|
||||
scope: text.frontmatter.description
|
||||
ignorecase: true
|
||||
raw:
|
||||
- '^This (skill|agent)\b'
|
||||
- '^This\b'
|
||||
|
||||
@@ -10,6 +10,10 @@ Additional documentation agents load on demand.
|
||||
|
||||
| File | Purpose |
|
||||
|------|---------|
|
||||
| `description-quality.md` | Qualitative guide for borderline description findings — action-verb rules, specificity criteria, proactive-use caveat, length limits. |
|
||||
| `field-inventory.md` | Canonical list of valid CC and Copilot agent definition fields. Load when the script needs authoritative field lists for structural validation. |
|
||||
| `description-quality.md` | Rubric for the description dimension — three-part shape, the 250/400-character budget, the hand-invoked contract, and the internal-mechanics FAIL. |
|
||||
| `body-and-delegation.md` | Rubric for the body, delegation and comment-discipline dimensions — the delegation FAIL, why agents take no body word gate, and what an agent body is for. |
|
||||
| `scope-plugin-apm.md` | Contract for a single vendor-neutral `.apm/agents/<name>.agent.md` file — allowlist, dimension routing, and the dimensions that do not apply. |
|
||||
| `scope-project-user.md` | Contract for a Claude Code / Copilot file pair — counterpart derivation, provider field rules, and pair consistency. |
|
||||
| `validation-scripts.md` | Loaded only when a Step 1 script fails or cannot run — scope-detection walk-up, manual fallback checks, and known script failures. |
|
||||
| `field-inventory.md` | Authoritative field lists, read as data by `validate.sh`: valid CC and Copilot agent fields, and the vendor-neutral plugin/APM allowlist. |
|
||||
| `sources.md` | Research provenance records for skill content. Load only when tracing the origin of a specific rule or field constraint. |
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
---
|
||||
source_keys:
|
||||
- context7-websites-code-claude
|
||||
- claude-code-plugins-docs
|
||||
- claude-code-subagents-docs
|
||||
- context7-github-en-copilot
|
||||
- github-custom-agents-configuration
|
||||
---
|
||||
|
||||
# Body, Delegation and Comment Discipline Reference
|
||||
|
||||
Upstream source: Claude Code subagent and plugin references, GitHub Copilot custom-agents
|
||||
configuration. House contract: ADR-0020, the context budget.
|
||||
|
||||
Read this when judging the **body**, **delegation** and **comment-discipline** dimensions.
|
||||
|
||||
## The core test
|
||||
|
||||
For every sentence in the body, ask: **"Would the agent get this wrong without this instruction?"**
|
||||
|
||||
If no — cut it. The agent already knows it from general training. Adding it wastes tokens and
|
||||
dilutes the signal of what matters.
|
||||
|
||||
## Agents take no body word gate
|
||||
|
||||
ADR-0020 gates a skill body at 600 words SUGGESTION / 900 FAIL and deliberately gates an agent body
|
||||
at nothing. The two are not the same construct: a skill body is loaded into the caller's live
|
||||
context and competes with the conversation already there, while an agent body *becomes* the system
|
||||
prompt of a fresh context that has nothing else in it. The rationale for the 900-word ceiling does
|
||||
not transfer, so:
|
||||
|
||||
- **Never report an agent body as too long on a word count.** There is no number to cite.
|
||||
- **Never add such a gate to `scripts/validate.sh`.** `tests/validate.bats` pins its absence with a
|
||||
body far past 900 words that must still pass, and adding one would contradict the ADR.
|
||||
- The one length signal that does apply is the Copilot runtime's 30,000-character body limit, which
|
||||
`validate.sh` already reports as a SUGGESTION because content past it is silently truncated.
|
||||
|
||||
Length is judged through the delegation check below instead, which is the defect a word count was
|
||||
standing in for anyway.
|
||||
|
||||
## The delegation check
|
||||
|
||||
A plugin-scope agent is a single `.apm/agents/<name>.agent.md` file with no sibling `references/`
|
||||
directory. It cannot progressively disclose to itself — it can only delegate to skills. So a
|
||||
procedure spelled out in an agent body that a skill the agent invokes already owns is not a
|
||||
shortcut: it is a second copy of that procedure, and the second copy drifts. This is the
|
||||
characteristic agent defect, the way a stale README row is the characteristic skill defect.
|
||||
|
||||
**An agent body that restates a procedure owned by a skill it can invoke is a FAIL.** The Fix is
|
||||
always the same shape: invoke `<skill>` instead.
|
||||
|
||||
How to apply it: for each procedural block in the body — a rule list, a numbered sequence, a
|
||||
constraint table — ask which skill owns that procedure. If the agent names that skill anywhere (its
|
||||
dispatch table, its routing prose, its frontmatter), the block is a restatement and the skill is
|
||||
already there to be invoked.
|
||||
|
||||
Worked example. The three `*-orchestrate` agents exist to compose domain skills — `git-orchestrate`
|
||||
(933 body words), `gitea-orchestrate` (1,199) and `apm-orchestrate` (1,080) — so any step they
|
||||
spell out that the composed skill already owns is the defect. `git-orchestrate:24-31` carries a
|
||||
"Hard rules" list (Conventional Commits types, atomic commits, never commit secrets, git trailers)
|
||||
that `git-commits` owns and that `git-orchestrate:44` routes to by name; `:39` concedes the point
|
||||
outright, noting the sub-skills "carry their own local copies of these rules". Two copies, one
|
||||
authority, and nothing keeping them in step.
|
||||
|
||||
What is **not** a finding under this rule, because no skill owns it:
|
||||
|
||||
- The dispatch table itself — which operation routes to which skill.
|
||||
- Safety gates the agent enforces before dispatching, and refusals it makes on its own authority.
|
||||
- The input contract and the structured output the agent's caller consumes.
|
||||
- Session state the agent carries across skill invocations.
|
||||
|
||||
## What the body is for
|
||||
|
||||
Include what the fresh context lacks:
|
||||
|
||||
- A direct role instruction opening the prompt: `You are a [role]. When invoked, [action].`
|
||||
- One bounded job, stated so the agent knows what it must refuse.
|
||||
- The dispatch, gates, inputs and outputs listed above.
|
||||
- **Error handling** — what the agent does on malformed, missing or contradictory input: stop and
|
||||
report, or degrade to a named fallback. Absent it, the agent invents a recovery, and a
|
||||
subagent's invented recovery is invisible to its caller until the output is wrong.
|
||||
- Non-obvious environment facts and project-specific conventions it cannot infer.
|
||||
- One default per decision point with one escape hatch.
|
||||
|
||||
Do not include at all:
|
||||
|
||||
- Concepts the agent already knows (what JSON is, how HTTP works, what a CSV is)
|
||||
- Exhaustive option lists — pick a default; the agent does not benefit from choosing
|
||||
- Steps the agent handles independently — over-specifying leads to unproductive paths
|
||||
- Restatements of the description, which is already in context
|
||||
|
||||
## Comment discipline
|
||||
|
||||
Inspect every comment block in the YAML frontmatter and apply the core test to each: *would the
|
||||
agent get this wrong without this comment?* Template scaffolding — `# Optional. <long
|
||||
explanation>`, more than a line or two of inline guidance per field — belongs to development, not
|
||||
to a shipped file. At plugin/APM scope the stakes are higher than tidiness: `apm compile` copies
|
||||
frontmatter verbatim to every target, `<!-- ... -->` is not valid YAML, and `validate.sh` FAILs a
|
||||
frontmatter block that still contains one.
|
||||
|
||||
## Auditing guidance
|
||||
|
||||
Flag as FAIL if:
|
||||
|
||||
- The body restates a procedure owned by a skill the agent can invoke — Fix: invoke `<skill>`
|
||||
instead
|
||||
- A sentence answers "no" to the core test — it is padding
|
||||
- A decision point presents a menu of options with no default
|
||||
- An instruction repeats content already in the description
|
||||
- Frontmatter comments are template scaffolding rather than instruction, or are HTML comments at
|
||||
plugin/APM scope
|
||||
- A prescriptive sequence is used where flexibility is fine, or the reverse
|
||||
|
||||
Flag as SUGGESTION if:
|
||||
|
||||
- The body does not open with a direct role instruction
|
||||
- The body specifies no error handling — nothing tells the agent what to do with malformed,
|
||||
missing or contradictory input
|
||||
- The job the agent describes is unbounded, or bounded only implicitly
|
||||
- A rationale is missing from a rule the agent is expected to enforce — present but unexplained
|
||||
- Comments are useful but verbose enough to bury the field they annotate
|
||||
@@ -1,7 +1,6 @@
|
||||
---
|
||||
source_keys:
|
||||
- context7-websites-code-claude
|
||||
- claude-code-plugins-docs
|
||||
- claude-code-subagents-docs
|
||||
- context7-github-en-copilot
|
||||
- github-custom-agents-configuration
|
||||
@@ -9,41 +8,123 @@ source_keys:
|
||||
|
||||
# Agent Description Quality Reference
|
||||
|
||||
Load this file when a description finding is borderline and you need to make a precise call.
|
||||
Upstream source: Claude Code subagent reference, GitHub Copilot custom-agents configuration.
|
||||
House contract: ADR-0020, the context budget. The house contract is narrower than either
|
||||
platform's schema rather than a reinterpretation of it: where both speak, both must be satisfied.
|
||||
|
||||
## Action-verb opening
|
||||
## Why the description is the expensive part
|
||||
|
||||
The description must open with an imperative or present-tense verb that describes what the agent does ("Reviews...", "Audits...", "Generates...", "Analyzes..."). Avoid:
|
||||
- Noun phrases: "An agent that..." — no verb
|
||||
- "This agent..." or "Use this when..." — passive framing
|
||||
- "Helps with..." — too vague to be a clear verb
|
||||
At startup an agent loads only the `name` and `description` of every installed skill and agent.
|
||||
The body is never seen until the agent is invoked. The description therefore carries the entire
|
||||
triggering burden **and** is paid for in every session, whether the agent fires or not.
|
||||
|
||||
**Borderline call:** "Validates and reviews..." is acceptable — two verbs is fine if both are specific. "Assists in reviewing..." is not — "assists" is vague filler.
|
||||
A second cost is less obvious and is a correctness hazard rather than a token cost: a description
|
||||
that summarises the workflow is a shortcut the caller takes *instead of* reading the body. A
|
||||
measured failure upstream — a description saying "code review between tasks" — produced one review
|
||||
where the body's flowchart specified two.
|
||||
|
||||
## Specificity of trigger condition
|
||||
## Step 0 — establish which contract applies
|
||||
|
||||
The description must state what specifically triggers the agent. Generic phrasing fails:
|
||||
- Too vague: "when the user needs help with agents"
|
||||
- Acceptable: "when the user says 'audit this agent', 'check if my agent follows best practices', or wants to know if an agent pair is ready to ship"
|
||||
Read the frontmatter before judging a single word.
|
||||
|
||||
Include indirect triggers: "even if they don't use the word 'audit'" or "even if the user doesn't phrase it as a review request". If the agent should activate on a recognisable user goal (not just literal keyword matches), name that goal.
|
||||
- **`disable-model-invocation: true` or `user-invocable: false`** — the agent is hand-invoked. Its
|
||||
description is never matched against user intent, so it is not a routing string. It carries **one
|
||||
plain human-facing sentence** stating what the agent does. Audit it for that and nothing else.
|
||||
Reporting a missing trigger clause, a missing boundary clause or absent indirect triggers on a
|
||||
hand-invoked agent is a wrong finding, not a strict one. Both fields are Copilot-only and neither
|
||||
is on the vendor-neutral APM allowlist, so this case arises in a Copilot `.agent.md` at
|
||||
project/user scope and nowhere else. Its Claude Code counterpart has no equivalent field and stays
|
||||
model-invoked, so the two halves of the pair carrying differently shaped descriptions is expected
|
||||
there rather than a pair-consistency finding.
|
||||
- **No such flag** — the agent is model-invoked and the rest of this file applies.
|
||||
|
||||
**Borderline call:** If the description covers direct triggers but omits common indirect phrasings that a user would plausibly use, mark as SUGGESTION (not FAIL) — the agent still activates, just less reliably.
|
||||
## The three-part shape
|
||||
|
||||
## `Use proactively`
|
||||
A model-invoked description carries exactly three things:
|
||||
|
||||
For CC files: including "Use proactively" signals the CC runtime to offer the agent unprompted when conditions are met. This is CC-specific — use it when the agent should activate without an explicit user request.
|
||||
1. **Trigger clause.** When to invoke, phrased imperatively: `Use when ...`. Not `This agent ...` —
|
||||
the caller is deciding whether to act, not reading a catalogue entry.
|
||||
2. **At most one capability clause.** What it does, in one clause. Never an enumeration.
|
||||
3. **Boundary clause.** Compressed form: `Not <thing> -> <skill-name>.` The target must resolve to
|
||||
a real skill directory or agent file in the authoring source.
|
||||
|
||||
For Copilot files: this phrase has no effect. Use `user-invocable: false` / `disable-model-invocation: true` for equivalent Copilot behavior. Flag `Use proactively` in a Copilot description as a SUGGESTION (not FAIL) — it causes no harm, just has no effect.
|
||||
Everything else belongs in the body or in the plugin's `README.md`.
|
||||
|
||||
## Length and hard limits
|
||||
## Indirect triggers — conditional, never blanket
|
||||
|
||||
- CC agent descriptions: no documented character limit, but keep under 500 characters to avoid truncation in UI contexts.
|
||||
- Copilot agent descriptions: no separate documented limit, but the overall 30,000-character body limit applies to the full file.
|
||||
- Skill descriptions (SKILL.md): hard 1024-character limit enforced by the platform.
|
||||
Add "even if the user doesn't say X" **only where the user's natural phrasing genuinely omits the
|
||||
domain word.** True for the `gitea-*` family: people say "create an issue", not "create a Gitea
|
||||
issue". False for `git-commits`: nobody asks for a commit without saying commit. A blanket
|
||||
indirect-trigger clause on an agent whose domain word is unavoidable is padding charged to every
|
||||
session.
|
||||
|
||||
## Do not use when
|
||||
## Near-miss exclusions
|
||||
|
||||
Include a "Do not use when..." clause only if a near-miss agent or skill exists that could steal activations. Omitting it is not a finding. Including it is correct when there is a real confusion risk (e.g., `/agent-audit` vs `/skill-audit`).
|
||||
Add a boundary clause only where a sibling skill or agent could plausibly steal the activation. Use
|
||||
strong near-misses — queries that share keywords but need something different — not weak ones. One
|
||||
boundary clause per genuine near-miss; a list of four is enumeration wearing a boundary's clothes.
|
||||
|
||||
**Borderline call:** If the "Do not use when" clause is present but the exclusion described is already obvious from context, mark as SUGGESTION to tighten or remove — not FAIL.
|
||||
## Before / after
|
||||
|
||||
```yaml
|
||||
# FAIL — a noun-phrase opener rather than a trigger, capability enumeration in
|
||||
# place of one capability clause, and no boundary clause at all, preloaded into
|
||||
# every session forever. (The live git-orchestrate description, 254 chars.)
|
||||
description: Orchestrates git workflow operations for other agents. Invoke when a
|
||||
caller needs a multi-step or destructive git operation (rebase, force-push, branch
|
||||
deletion) coordinated across domain skills with safety gates, session context, and
|
||||
structured results.
|
||||
|
||||
# PASS — trigger, one capability clause, boundary. The operation list and the
|
||||
# safety-gate mechanics are the body's job; the router cannot act on them.
|
||||
description: >
|
||||
Use when an agent caller needs a multi-step or destructive git operation
|
||||
dispatched and safety-gated. Not conversational git help -> git-workflow.
|
||||
```
|
||||
|
||||
## Auditing guidance
|
||||
|
||||
Flag as FAIL if:
|
||||
|
||||
- **Over 400 characters.** Measured on the folded YAML value, not the raw source lines.
|
||||
`validate.sh` reports the number; do not re-derive it, but do point the Fix at what to cut. Agent
|
||||
descriptions have no platform-documented ceiling of their own — unlike a skill's 1,024-character
|
||||
spec limit, the 400-character house ceiling is the only hard limit there is, so do not go looking
|
||||
for a backstop behind it.
|
||||
- **Internal mechanics appear in the description.** Any of:
|
||||
- capability enumeration or a feature list;
|
||||
- output-format detail ("Produces a compact findings report with Why and Fix per finding");
|
||||
- composition or architecture notes ("composes X rather than duplicating Y", "a cross-cutting
|
||||
shared agent", "the human-facing entry point", "replaces the old flat invocation");
|
||||
- implementation detail ("self-validates via a bundled deterministic script").
|
||||
|
||||
None of it can change a routing decision and all of it is preloaded.
|
||||
`Kyberforge.CompositionNote` catches the common phrasings deterministically; the rest is
|
||||
judgment. This is the rule that deflates a description, so apply it before reaching for length.
|
||||
- **The same trigger stated twice in two registers** — a verb list, then the same verbs re-quoted
|
||||
as user phrasings, usually in the same order. One register, whichever routes better.
|
||||
- **Descriptive rather than imperative phrasing** (`This agent ...`, `This is the ...`).
|
||||
`Kyberforge.DescriptionOpener` catches any opener matching `^This`. There is no action-verb rule
|
||||
here and never was a defensible one: an `Orchestrates ...` or `Audits ...` opener is a catalogue
|
||||
entry, not a trigger.
|
||||
- **Vague capabilities** ("helps with agents" where "audits an agent definition pair" was
|
||||
available). `Kyberforge.VagueWording` catches the known filler; imprecision outside that list is
|
||||
judgment.
|
||||
- **A boundary clause naming a target that does not resolve** to a real skill directory or agent
|
||||
file in the authoring source. `validate.sh` resolves this for agent files at both scopes and
|
||||
reports each unresolved target itself — take its verdict rather than re-resolving the name by
|
||||
hand, because a hand-walk over a different universe can contradict it. What is left to you is
|
||||
semantic and the script cannot reach it: whether a target that *does* resolve is the right
|
||||
sibling to exclude, and whether a clause naming no target at all ("examine the files manually")
|
||||
should have named one.
|
||||
- **`Use proactively` in a Copilot or vendor-neutral description.**
|
||||
`KyberforgeCopilot.ProactivePhrase` catches it. The phrase steers the Claude Code runtime and
|
||||
does nothing anywhere else, so in a `.agent.md` it is preloaded text that buys no behaviour.
|
||||
- **Trigger-list, boundary or indirect-trigger content on a hand-invoked agent** — see Step 0.
|
||||
|
||||
Flag as SUGGESTION if:
|
||||
|
||||
- **Over 250 characters** but at or under 400. This tier is what moves the corpus average; the FAIL
|
||||
tier only stops outliers. Report it rather than treating a 399-character description as clean.
|
||||
- A near-miss exclusion is present but targets a weak near-miss.
|
||||
- An indirect trigger is present and warranted but could name the omitted phrasing more precisely.
|
||||
|
||||
@@ -25,4 +25,25 @@ target disable-model-invocation user-invocable mcp-servers metadata
|
||||
|
||||
## apm-agent-allowlist
|
||||
|
||||
name description model source_keys
|
||||
name description model source_keys disallowedTools
|
||||
|
||||
Parsing note: `validate.sh` reads the **first** non-empty, non-`#`, non-`---` line under each
|
||||
heading as a whitespace-separated token list, and stops there. Keep the token line immediately
|
||||
below its heading; explanatory prose goes after it, as here.
|
||||
|
||||
Why `disallowedTools` is on a list that is otherwise vendor-neutral, when `tools` is not
|
||||
(ADR-0016 and its 2026-08-14 amendment): the two are not symmetric. `tools` is an **allowlist**
|
||||
whose vocabulary differs per harness — Claude Code names its own tools, Copilot CLI uses aliases
|
||||
(`execute`/`read`/`edit`/`search`/`agent`/`web`) — so a value correct for one is wrong for the
|
||||
other, and `apm compile` copies frontmatter verbatim with no per-target integrator to reconcile
|
||||
them. `disallowedTools` is a **denylist**, and denying by name is safe under verbatim copy: a name
|
||||
the other harness does not recognise denies nothing, so the worst case is that the fence is absent
|
||||
there, never that the wrong capability is granted. Claude Code honours it for plugin subagents —
|
||||
`docs/research/docs/claude-code-plugins/agent-definition.md:99` names the fields plugin agents
|
||||
silently ignore (`hooks`, `mcpServers`, `permissionMode`) and `disallowedTools` is not among them.
|
||||
|
||||
`disallowedTools` also appears in `claude-code-only-fields` above, and that stays correct: at
|
||||
project/user scope it is still a Claude-only field and must not appear in a Copilot `.agent.md`.
|
||||
The two lists answer different questions — "may this field cross the CC/Copilot file boundary" for
|
||||
a real pair, versus "is this field safe under verbatim copy to every target" for a single
|
||||
vendor-neutral APM file.
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
---
|
||||
source_keys:
|
||||
- claude-code-plugins-docs
|
||||
- claude-code-subagents-docs
|
||||
- github-custom-agents-configuration
|
||||
---
|
||||
|
||||
# Plugin/APM Scope Contract
|
||||
|
||||
Read this when the agent file sits at `<package>/.apm/agents/<name>.agent.md` — a single
|
||||
vendor-neutral file inside an APM package, with no counterpart anywhere.
|
||||
|
||||
## What is different here
|
||||
|
||||
`apm compile` copies an agent's frontmatter **verbatim** to every target harness. There is no
|
||||
per-target integrator to reconcile a Claude-Code-only field with a Copilot-only one, so the file
|
||||
cannot carry either (ADR-0016). That single fact drives everything below.
|
||||
|
||||
## Frontmatter allowlist
|
||||
|
||||
The permitted keys are the `apm-agent-allowlist` section of `references/field-inventory.md`. Read
|
||||
them from there. Do not recite the list in a finding, do not work from memory, and do not trust any
|
||||
restatement of it you find elsewhere in this repo: the list is data with one home (ADR-0009), it
|
||||
has changed before, and `validate.sh` parses that same section at load time, so a recitation is a
|
||||
copy that can disagree with the check the agent just ran.
|
||||
|
||||
`field-inventory.md` records why a denylist-shaped field is admitted where an allowlist-shaped one
|
||||
is not. Read that note before arguing with a finding about it.
|
||||
|
||||
## Dimension routing
|
||||
|
||||
`validate.sh` findings land as follows at this scope:
|
||||
|
||||
| Finding | Dimension |
|
||||
|---|---|
|
||||
| any frontmatter key outside the allowlist; body over the 30,000-character Copilot limit | Provider safety |
|
||||
| everything else — missing or malformed field, `name` not matching the filename stem, empty body, absent frontmatter, template HTML comments, description length | Structure |
|
||||
| — | Pair consistency never applies |
|
||||
|
||||
**Provider safety means something else here.** At project/user scope it asks whether a field leaked
|
||||
across the Claude Code / Copilot boundary. At this scope there is no boundary and no pair: it asks
|
||||
whether every field survives a verbatim copy to *every* target. Report it in those terms — a
|
||||
finding phrased as "CC-only field in a Copilot file" is the wrong finding here.
|
||||
|
||||
**Pair consistency never applies.** There is one file by design. `validate.sh` never emits a
|
||||
missing-counterpart FAIL at this scope, and neither do you, under any circumstance. Drop
|
||||
`pair-consistency` from the Step 4 coverage line rather than reporting it clean.
|
||||
|
||||
## Behaviour the schema cannot express
|
||||
|
||||
Read the description and body. If either implies a need the vendor-neutral frontmatter can no
|
||||
longer express — a tool restriction, `isolation`, `memory`, or another Claude-only behaviour a
|
||||
hand-authored CC file could have declared — flag it as a **SUGGESTION, never a FAIL**. This is a
|
||||
known upstream schema limitation (ADR-0016), not an authoring mistake, and the finding exists to
|
||||
give the author visibility into the gap rather than to imply the schema can be made to close it.
|
||||
|
||||
Example: a body saying "only use Read and Grep, never Edit" with no `tools` field to enforce it.
|
||||
A denylist-shaped restriction is the available half of that — see `field-inventory.md`.
|
||||
@@ -0,0 +1,59 @@
|
||||
---
|
||||
source_keys:
|
||||
- context7-websites-code-claude
|
||||
- claude-code-subagents-docs
|
||||
- context7-github-en-copilot
|
||||
- github-custom-agents-configuration
|
||||
---
|
||||
|
||||
# Project and User Scope Contract
|
||||
|
||||
Read this when the agent file is not under `.apm/agents/` — a Claude Code `.md` and Copilot CLI
|
||||
`.agent.md` **pair**, at project scope (`<repo>/.claude/agents/` and `<repo>/.github/agents/`) or
|
||||
user scope (`~/.claude/agents/` and `~/.copilot/agents/`). `validate.sh` derives the counterpart
|
||||
from whichever half it was handed; audit both.
|
||||
|
||||
## The pair is a house convention
|
||||
|
||||
Neither platform requires a counterpart file. The pair is a kyberforge convention (ADR-0005), so a
|
||||
missing counterpart is a FAIL against **this repo's** convention and must be labelled that way in
|
||||
the finding, not presented as a platform spec failure.
|
||||
|
||||
## Dimension routing
|
||||
|
||||
`validate.sh` findings land as follows at these scopes:
|
||||
|
||||
| Finding | Dimension |
|
||||
|---|---|
|
||||
| a Claude-Code-only field in the Copilot file, a Copilot-only field in the CC file, a tool the runtime withholds from subagents, body over the 30,000-character Copilot limit | Provider safety |
|
||||
| counterpart file not found | Pair consistency |
|
||||
| everything else — missing or malformed field, name format, empty body, absent frontmatter, description length | Structure |
|
||||
|
||||
The two field lists are the `claude-code-only-fields` and `copilot-only-fields` sections of
|
||||
`references/field-inventory.md`. Read them from there rather than from memory; `validate.sh` parses
|
||||
those same sections, so any restatement is a copy that can disagree with the check (ADR-0009).
|
||||
|
||||
## Field and naming rules that differ by provider
|
||||
|
||||
- `name` must match the filename stem in a **Copilot CLI** `.agent.md`. Claude Code imposes no such
|
||||
rule, so a CC file whose `name` differs from its filename is not a finding.
|
||||
- A Copilot **cloud/IDE** agent — one under `.github/copilot/agents/` — may omit `name` entirely.
|
||||
If it carries one, it still has to be kebab-case.
|
||||
- `Use proactively` is meaningful in a CC description and steers the runtime to offer the agent
|
||||
unprompted. In a Copilot description it does nothing; `KyberforgeCopilot.ProactivePhrase` flags
|
||||
it. The Copilot equivalent is `disable-model-invocation` / `user-invocable`, which changes the
|
||||
description contract entirely — see `references/description-quality.md`, Step 0.
|
||||
|
||||
## Pair consistency
|
||||
|
||||
Check that:
|
||||
|
||||
- Both files exist.
|
||||
- Both system prompt bodies are non-empty (`validate.sh` covers this; do it by hand only when the
|
||||
script could not run).
|
||||
- The two files describe the **same job**. Divergent capability claims across the pair mean one
|
||||
half was edited and the other was not, which is the defect this dimension exists to catch.
|
||||
- Descriptions may legitimately differ in *shape* when the Copilot half is hand-invoked — that is
|
||||
the Step 0 case in `references/description-quality.md`, not a pair-consistency finding.
|
||||
|
||||
Keep `pair-consistency` in the Step 4 coverage line at these scopes.
|
||||
@@ -14,7 +14,7 @@ source_keys:
|
||||
- **URL:** context7:/websites/code_claude
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/claude-code-plugins/sources.md
|
||||
- **Description:** Official Claude Code documentation site indexed by Context7 — plugin manifest schema, subagent definition types, marketplace JSON format, agent markdown file format
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/description-quality.md
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/description-quality.md, references/body-and-delegation.md, references/scope-project-user.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## claude-code-plugins-docs
|
||||
@@ -22,7 +22,7 @@ source_keys:
|
||||
- **URL:** https://code.claude.com/docs/en/plugins
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/claude-code-plugins/sources.md
|
||||
- **Description:** Official Claude Code plugin authoring guide — plugin structure, manifest fields, loading methods, skill namespacing, agent activation, marketplace submission
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/description-quality.md
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/body-and-delegation.md, references/scope-plugin-apm.md, references/validation-scripts.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## claude-code-subagents-docs
|
||||
@@ -30,7 +30,7 @@ source_keys:
|
||||
- **URL:** https://code.claude.com/docs/en/sub-agents
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/claude-code-plugins/sources.md
|
||||
- **Description:** Official Claude Code subagent reference — definition format, all frontmatter fields, scope priority, built-in agents, CLI flags, environment variables, known limitations
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/description-quality.md
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/description-quality.md, references/body-and-delegation.md, references/scope-plugin-apm.md, references/scope-project-user.md, references/validation-scripts.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## context7-github-en-copilot
|
||||
@@ -38,7 +38,7 @@ source_keys:
|
||||
- **URL:** context7:/websites/github_en_copilot
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/github-copilot-plugins/sources.md
|
||||
- **Description:** Official GitHub Copilot documentation indexed by Context7; covers CLI plugins, custom agents, SDK, and marketplace
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/description-quality.md
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/description-quality.md, references/body-and-delegation.md, references/scope-project-user.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## github-custom-agents-configuration
|
||||
@@ -46,7 +46,7 @@ source_keys:
|
||||
- **URL:** https://docs.github.com/en/copilot/reference/custom-agents-configuration
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/github-copilot-plugins/sources.md
|
||||
- **Description:** Reference for cloud and IDE custom agent definition format — frontmatter fields, tool aliases, MCP server config, secrets interpolation, scoping hierarchy
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/description-quality.md
|
||||
- **Contributing files:** SKILL.md, references/field-inventory.md, references/description-quality.md, references/body-and-delegation.md, references/scope-plugin-apm.md, references/scope-project-user.md, references/validation-scripts.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## github-cli-plugin-reference
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
---
|
||||
source_keys:
|
||||
- claude-code-plugins-docs
|
||||
- claude-code-subagents-docs
|
||||
- github-custom-agents-configuration
|
||||
---
|
||||
|
||||
# Validation Scripts Reference
|
||||
|
||||
Read this when a Step 1 script fails, cannot run, or reports something that needs interpreting.
|
||||
Nothing here is needed on a clean run.
|
||||
|
||||
## Report the gap, do not guess
|
||||
|
||||
If a script cannot run at all — Bash denied, `python3` unavailable, `vale` not installed — say so
|
||||
as an **INFO** finding naming the script and the missing dependency, then fall back to the manual
|
||||
checks below. An INFO never changes PASS/FAIL. Silently omitting the dimension a script would have
|
||||
covered reports a clean audit that checked less than it claims to have checked.
|
||||
|
||||
## How the scripts detect scope
|
||||
|
||||
`validate.sh` and `validate-provenance.sh` walk up from the agent file's directory and stop at the
|
||||
first of these:
|
||||
|
||||
1. An `apm.yml` carrying a top-level `type: instructions|skill|hybrid|prompts` line — **plugin/APM
|
||||
scope**, and that directory is the package root. An `apm.yml` with no `type:` is a
|
||||
marketplace-only manifest: skip it and keep walking.
|
||||
2. `$HOME` — **user scope**, checked before `.git` so a dotfiles-managed home directory that is its
|
||||
own repo cannot shadow it.
|
||||
3. A `.git` directory or file — **project scope**.
|
||||
4. The filesystem root — **project scope**.
|
||||
|
||||
`plugin.json` and `.claude-plugin/plugin.json` are not scope signals. A directory holding only a
|
||||
`plugin.json` and no `apm.yml` falls through to project or user scope.
|
||||
|
||||
`validate-provenance.sh` exits 0 silently when that walk does not land on a package root, and again
|
||||
when the package has no provenance data. Silence from it is a pass, not a skip you need to
|
||||
investigate.
|
||||
|
||||
## Manual fallback
|
||||
|
||||
**Every scope:** required fields present (`name`, `description`, non-empty body); `name` is
|
||||
kebab-case; no `FILL IN:` placeholders in the description or body; the description at or under 400
|
||||
characters measured on the folded YAML value.
|
||||
|
||||
**Plugin/APM scope:** `name` matches the filename stem; no HTML comments left in the frontmatter;
|
||||
no frontmatter key outside the `apm-agent-allowlist` section of `references/field-inventory.md` —
|
||||
open that file, do not work from memory.
|
||||
|
||||
**Project/user scope:** the counterpart file exists; `name` matches the filename stem in the
|
||||
Copilot `.agent.md` only (Claude Code files are exempt); no key from `claude-code-only-fields` in
|
||||
the Copilot file and none from `copilot-only-fields` in the CC file, both read from
|
||||
`references/field-inventory.md`.
|
||||
|
||||
## Script-specific failures
|
||||
|
||||
- **`Error: field-inventory.md not found` (exit 2).** `validate.sh` reads its field lists from
|
||||
`references/field-inventory.md` at load time and refuses to run without it, rather than falling
|
||||
back to a hardcoded list that could disagree with the file (ADR-0009). Restore the file; do not
|
||||
work around it.
|
||||
- **`vale` reports `0 files`.** Treat the pass as NOT RUN, not as clean, and fall back to full
|
||||
Step 3 judgment for the dimensions it would have covered. The `Kyberforge` style is scoped to
|
||||
`**/agents/*.md` and `**/*.agent.md`, and `KyberforgeCopilot` to `**/*.agent.md` alone — a file
|
||||
outside those globs is silently not linted.
|
||||
- **`E100 Runtime error ... does not exist` (exit 2) from `vale-wrap.sh`.** An explicit relative
|
||||
`--config` was passed. Pass none: the wrapper locates its own `assets/vale/.vale.ini` from its
|
||||
own path. Do not read this exit code as vale being unavailable.
|
||||
- **A path argument that does not exist is a hard error** in `vale-wrap.sh`, deliberately: bare
|
||||
`vale` would fall back to reading stdin and print a clean-looking `0 errors ... in stdin`, which
|
||||
the `0 files` guard above does not catch.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -31,6 +31,44 @@ description: A valid agent description.
|
||||
${extra_frontmatter}
|
||||
---
|
||||
|
||||
You are a test agent. When invoked, do the thing.
|
||||
EOF
|
||||
}
|
||||
|
||||
# Helper: a description of EXACTLY <n> characters that carries a boundary
|
||||
# clause and names no routing target. ADR-0020's missing-boundary-clause
|
||||
# SUGGESTION fires on any description without one, so a fixture that omits it
|
||||
# is never "otherwise clean" and a test refuting SUGGESTION would be asserting
|
||||
# the boundary check's absence instead of the thing it names. The clause is
|
||||
# paid for out of the measured budget rather than appended to it, because
|
||||
# these tests measure the description LENGTH. "anything else" is not
|
||||
# hyphenated, so no routing target comes with it.
|
||||
desc_of_length() {
|
||||
python3 - "$1" <<'PY'
|
||||
import sys
|
||||
n = int(sys.argv[1])
|
||||
prefix = 'Use when doing the thing. Do not use for anything else. '
|
||||
assert n >= len(prefix), 'requested description shorter than the boundary clause'
|
||||
print(prefix + 'x' * (n - len(prefix)))
|
||||
PY
|
||||
}
|
||||
|
||||
# Helper: same shape as make_apm_agent, but the description is supplied
|
||||
# verbatim — used by the ADR-0020 description-budget tests.
|
||||
make_apm_agent_with_desc() {
|
||||
local root="$1" name="$2" desc="$3"
|
||||
mkdir -p "$root/.apm/agents"
|
||||
cat > "$root/apm.yml" <<EOF
|
||||
name: test-package
|
||||
version: 0.1.0
|
||||
type: skill
|
||||
EOF
|
||||
cat > "$root/.apm/agents/${name}.agent.md" <<EOF
|
||||
---
|
||||
name: ${name}
|
||||
description: ${desc}
|
||||
---
|
||||
|
||||
You are a test agent. When invoked, do the thing.
|
||||
EOF
|
||||
}
|
||||
@@ -607,6 +645,139 @@ EOF
|
||||
refute_output --partial "FAIL"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ADR-0020 — description budget (250 SUGGESTION / 400 FAIL)
|
||||
#
|
||||
# Agents take the SAME description gates as skills: name + description is
|
||||
# preloaded into every session identically. Agents take NO body word gate — see
|
||||
# the final test in this block, which pins that asymmetry.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@test "ADR-0020: agent description of exactly 250 chars raises no suggestion" {
|
||||
local root="$TMPDIR/pkg"
|
||||
make_apm_agent_with_desc "$root" "my-agent" "$(desc_of_length 250)"
|
||||
run bash "$SCRIPT" "$root/.apm/agents/my-agent.agent.md"
|
||||
assert_success
|
||||
refute_output --partial "SUGGESTION"
|
||||
}
|
||||
|
||||
@test "ADR-0020: agent description of 251 chars raises a SUGGESTION and still exits 0" {
|
||||
local root="$TMPDIR/pkg"
|
||||
make_apm_agent_with_desc "$root" "my-agent" "$(desc_of_length 251)"
|
||||
run bash "$SCRIPT" "$root/.apm/agents/my-agent.agent.md"
|
||||
assert_success
|
||||
assert_output --partial "SUGGESTION"
|
||||
assert_output --partial "description is 251 chars"
|
||||
}
|
||||
|
||||
@test "ADR-0020: agent description of exactly 400 chars is a SUGGESTION, not a FAIL" {
|
||||
local root="$TMPDIR/pkg"
|
||||
make_apm_agent_with_desc "$root" "my-agent" "$(desc_of_length 400)"
|
||||
run bash "$SCRIPT" "$root/.apm/agents/my-agent.agent.md"
|
||||
assert_success
|
||||
assert_output --partial "SUGGESTION"
|
||||
}
|
||||
|
||||
@test "ADR-0020: agent description of 401 chars FAILs and exits non-zero" {
|
||||
local root="$TMPDIR/pkg"
|
||||
make_apm_agent_with_desc "$root" "my-agent" "$(desc_of_length 401)"
|
||||
run bash "$SCRIPT" "$root/.apm/agents/my-agent.agent.md"
|
||||
assert_failure
|
||||
assert_output --partial "description is 401 chars"
|
||||
assert_output --partial "400-character ADR-0020 ceiling"
|
||||
}
|
||||
|
||||
@test "ADR-0020: agent description length is measured after YAML folding is resolved" {
|
||||
local root="$TMPDIR/pkg"
|
||||
mkdir -p "$root/.apm/agents"
|
||||
cat > "$root/apm.yml" <<EOF
|
||||
name: test-package
|
||||
version: 0.1.0
|
||||
type: skill
|
||||
EOF
|
||||
# 11 folded lines of 40 chars + 10 joining spaces = 450 characters. Read off
|
||||
# the raw `description: >` line it is 1 character and passes.
|
||||
{
|
||||
echo "---"
|
||||
echo "name: my-agent"
|
||||
echo "description: >"
|
||||
python3 -c "print('\n'.join([' ' + 'x' * 40] * 11))"
|
||||
echo "---"
|
||||
echo ""
|
||||
echo "You are a test agent."
|
||||
} > "$root/.apm/agents/my-agent.agent.md"
|
||||
run bash "$SCRIPT" "$root/.apm/agents/my-agent.agent.md"
|
||||
assert_failure
|
||||
assert_output --partial "description is 450 chars"
|
||||
}
|
||||
|
||||
@test "ADR-0020: the description gate applies at project scope too" {
|
||||
local root="$TMPDIR/project"
|
||||
local desc
|
||||
desc="$(python3 -c "print('x' * 401)")"
|
||||
mkdir -p "$root/.git" "$root/.claude/agents" "$root/.github/agents"
|
||||
cat > "$root/.claude/agents/my-agent.md" <<EOF
|
||||
---
|
||||
name: my-agent
|
||||
description: $desc
|
||||
---
|
||||
|
||||
You are a test agent. When invoked, do the thing.
|
||||
EOF
|
||||
cat > "$root/.github/agents/my-agent.agent.md" <<EOF
|
||||
---
|
||||
name: my-agent
|
||||
description: A valid agent description.
|
||||
---
|
||||
|
||||
You are a test agent. When invoked, do the thing.
|
||||
EOF
|
||||
run bash "$SCRIPT" "$root/.claude/agents/my-agent.md"
|
||||
assert_failure
|
||||
assert_output --partial "400-character ADR-0020 ceiling"
|
||||
}
|
||||
|
||||
@test "ADR-0020: agents take NO body word gate — a body far over the 900-word skill ceiling passes" {
|
||||
local root="$TMPDIR/pkg"
|
||||
mkdir -p "$root/.apm/agents"
|
||||
cat > "$root/apm.yml" <<EOF
|
||||
name: test-package
|
||||
version: 0.1.0
|
||||
type: skill
|
||||
EOF
|
||||
# Deliberate asymmetry, not an oversight: a skill body is loaded into the
|
||||
# caller's context and competes with the live conversation, while an agent
|
||||
# body becomes the system prompt of a fresh context. ADR-0020 gates the
|
||||
# former at 900 words and explicitly declines to gate the latter. If a body
|
||||
# word gate is ever added here, it contradicts the ADR.
|
||||
#
|
||||
# The description carries a boundary clause so the ONLY thing this test can
|
||||
# go red on is a body finding. Without one, the missing-boundary-clause
|
||||
# SUGGESTION fires and the blanket `refute_output --partial "SUGGESTION"`
|
||||
# below trips for a reason that has nothing to do with body length — which
|
||||
# would look like the invariant breaking while proving nothing about it.
|
||||
# AGENTS.md cites this test as the pin for that invariant, so it has to fail
|
||||
# for one reason and one reason only.
|
||||
{
|
||||
echo "---"
|
||||
echo "name: my-agent"
|
||||
echo "description: A valid agent description. Do not use for anything else."
|
||||
echo "---"
|
||||
echo ""
|
||||
python3 -c "print(' '.join(['word'] * 1500))"
|
||||
} > "$root/.apm/agents/my-agent.agent.md"
|
||||
run bash "$SCRIPT" "$root/.apm/agents/my-agent.agent.md"
|
||||
assert_success
|
||||
refute_output --partial "FAIL"
|
||||
# A 1,500-word body is 667% of the skill ceiling. Nothing may be said about
|
||||
# it at any tier: not a FAIL, not a SUGGESTION, and not the word-count
|
||||
# wording either tier would use if a gate were quietly added later.
|
||||
refute_output --partial "SUGGESTION"
|
||||
refute_output --partial "1500 words"
|
||||
refute_output --partial "900-word"
|
||||
refute_output --partial "body is"
|
||||
}
|
||||
|
||||
@test "a bare plugin.json with no apm.yml is no longer plugin scope — falls through to project scope" {
|
||||
local root="$TMPDIR/proj-legacy-plugin-json"
|
||||
mkdir -p "$root/.git" "$root/.claude/agents" "$root/.github/agents"
|
||||
@@ -639,3 +810,136 @@ EOF
|
||||
assert_success
|
||||
refute_output --partial "hooks"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# tools: — both YAML spellings
|
||||
# ---------------------------------------------------------------------------
|
||||
# The subagent-unavailable-tool SUGGESTION is read off the `tools` field, and
|
||||
# `tools` has two legal spellings: an inline scalar and a block sequence. The
|
||||
# field used to be pulled out with a line regex whose capture is newline-bounded
|
||||
# on purpose, so a block sequence captured NOTHING and the check silently
|
||||
# stopped firing — on the shape Copilot agent files actually use, which is to say
|
||||
# on the files it was written for. Both spellings are pinned, and they are pinned
|
||||
# together: the inline case alone was green throughout.
|
||||
|
||||
# make_pair <root> <tools-frontmatter> — a project-scope CC + Copilot pair
|
||||
# carrying the same `tools` value in both files. `tools` is on neither the
|
||||
# claude-code-only nor the copilot-only list, so it is legal in both and the pair
|
||||
# stays otherwise clean; the description carries a boundary clause so the only
|
||||
# SUGGESTION that can fire is the one under test.
|
||||
make_tools_pair() {
|
||||
local root="$1" tools="$2"
|
||||
mkdir -p "$root/.git" "$root/.claude/agents" "$root/.github/agents"
|
||||
local f
|
||||
for f in "$root/.claude/agents/my-agent.md" "$root/.github/agents/my-agent.agent.md"; do
|
||||
{
|
||||
echo "---"
|
||||
echo "name: my-agent"
|
||||
echo "description: A valid agent description. Do not use for anything else."
|
||||
echo "$tools"
|
||||
echo "---"
|
||||
echo ""
|
||||
echo "You are a test agent. When invoked, do the thing."
|
||||
} > "$f"
|
||||
done
|
||||
}
|
||||
|
||||
@test "a subagent-unavailable tool in an INLINE tools scalar raises a SUGGESTION" {
|
||||
make_tools_pair "$TMPDIR/inline" "tools: Read ExitPlanMode"
|
||||
run bash "$SCRIPT" "$TMPDIR/inline/.claude/agents/my-agent.md"
|
||||
assert_success
|
||||
assert_output --partial "'ExitPlanMode' is listed in tools but is never available to subagents"
|
||||
}
|
||||
|
||||
@test "a subagent-unavailable tool in a BLOCK SEQUENCE tools field raises the same SUGGESTION" {
|
||||
make_tools_pair "$TMPDIR/block" "$(printf 'tools:\n - Read\n - ExitPlanMode')"
|
||||
run bash "$SCRIPT" "$TMPDIR/block/.claude/agents/my-agent.md"
|
||||
assert_success
|
||||
assert_output --partial "'ExitPlanMode' is listed in tools but is never available to subagents"
|
||||
}
|
||||
|
||||
@test "a tools list with no subagent-unavailable tool stays silent in both spellings" {
|
||||
# The control. Without it both cases above are satisfied by a check that
|
||||
# fires on every tools field it can see, which would be the opposite defect.
|
||||
make_tools_pair "$TMPDIR/inline-clean" "tools: Read Edit"
|
||||
run bash "$SCRIPT" "$TMPDIR/inline-clean/.claude/agents/my-agent.md"
|
||||
assert_success
|
||||
refute_output --partial "never available to subagents"
|
||||
|
||||
make_tools_pair "$TMPDIR/block-clean" "$(printf 'tools:\n - Read\n - Edit')"
|
||||
run bash "$SCRIPT" "$TMPDIR/block-clean/.claude/agents/my-agent.md"
|
||||
assert_success
|
||||
refute_output --partial "never available to subagents"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# A file that cannot be read
|
||||
# ---------------------------------------------------------------------------
|
||||
# scripts/check-apm-agents-valid.sh derives its expected agent-file set from
|
||||
# `git ls-files`, so it hands this script paths that are tracked but absent from
|
||||
# the worktree — a real and expected state, not a corner case. That used to exit
|
||||
# 1 with a bare FileNotFoundError traceback and no FAIL line at all: non-zero, so
|
||||
# the gate blocked, but with an interpreter stack instead of a diagnostic naming
|
||||
# the file. Both scope paths are covered because they are separate call sites
|
||||
# (check_apm_agent_file and check_file) and each needed its own handler.
|
||||
#
|
||||
# `is-a-dir.agent.md` is a DIRECTORY rather than a chmod 000 file on purpose:
|
||||
# these tests run as root in CI, where mode bits do not deny anything and a
|
||||
# permissions fixture would be silently readable and prove nothing.
|
||||
|
||||
@test "a nonexistent plugin/APM-scope agent file gets a FAIL naming the path, not a traceback" {
|
||||
local root="$TMPDIR/pkg"
|
||||
mkdir -p "$root/.apm/agents"
|
||||
cat > "$root/apm.yml" <<EOF
|
||||
name: test-package
|
||||
version: 0.1.0
|
||||
type: skill
|
||||
EOF
|
||||
run bash "$SCRIPT" "$root/.apm/agents/absent.agent.md"
|
||||
assert_failure
|
||||
assert_output --partial "FAIL"
|
||||
assert_output --partial "could not be read"
|
||||
assert_output --partial "absent.agent.md"
|
||||
refute_output --partial "Traceback"
|
||||
refute_output --partial "FileNotFoundError"
|
||||
}
|
||||
|
||||
@test "an unreadable plugin/APM-scope agent file gets a FAIL naming the path, not a traceback" {
|
||||
local root="$TMPDIR/pkg-dir"
|
||||
mkdir -p "$root/.apm/agents/is-a-dir.agent.md"
|
||||
cat > "$root/apm.yml" <<EOF
|
||||
name: test-package
|
||||
version: 0.1.0
|
||||
type: skill
|
||||
EOF
|
||||
run bash "$SCRIPT" "$root/.apm/agents/is-a-dir.agent.md"
|
||||
assert_failure
|
||||
assert_output --partial "FAIL"
|
||||
assert_output --partial "could not be read"
|
||||
assert_output --partial "is-a-dir.agent.md"
|
||||
refute_output --partial "Traceback"
|
||||
refute_output --partial "IsADirectoryError"
|
||||
}
|
||||
|
||||
@test "a nonexistent project-scope agent file gets a FAIL naming the path, not a traceback" {
|
||||
# The counterpart is pre-checked before either file is opened, so this
|
||||
# exercises the OTHER call site: the counterpart exists, the named file does
|
||||
# not, and check_file is what has to report it.
|
||||
local root="$TMPDIR/proj-missing"
|
||||
mkdir -p "$root/.git" "$root/.claude/agents" "$root/.github/agents"
|
||||
cat > "$root/.github/agents/my-agent.agent.md" <<EOF
|
||||
---
|
||||
name: my-agent
|
||||
description: A valid agent description. Do not use for anything else.
|
||||
---
|
||||
|
||||
You are a test agent. When invoked, do the thing.
|
||||
EOF
|
||||
run bash "$SCRIPT" "$root/.claude/agents/my-agent.md"
|
||||
assert_failure
|
||||
assert_output --partial "FAIL"
|
||||
assert_output --partial "could not be read"
|
||||
assert_output --partial "my-agent.md"
|
||||
refute_output --partial "Traceback"
|
||||
refute_output --partial "FileNotFoundError"
|
||||
}
|
||||
|
||||
@@ -30,16 +30,29 @@ bash scripts/new-agent.sh security-reviewer ~
|
||||
|
||||
| File | Purpose |
|
||||
|------|---------|
|
||||
| `SKILL.md` | Skill instructions for agents |
|
||||
| `SKILL.md` | Skill instructions for agents — gotchas, the create/improve dispatch table, the scope dispatch table, the shared gates, and validation/close |
|
||||
| `scripts/new-agent.sh` | Scaffolds agent definition file(s) from templates — a single `.apm/agents/<name>.agent.md` at plugin/APM scope, or a Claude Code + Copilot CLI pair at project/user scope |
|
||||
| `references/deployment-modes.md` | Plugin/APM vs project vs user scope: restrictions, scoped identifiers, path conventions |
|
||||
| `references/scripts.md` | Conventions for new-agent.sh and any future scripts: contract, template variables, file placement, error messages |
|
||||
| `references/create.md` | Create flow: prerequisites, scaffold and scope walk-up, what to fill in, package-root `sources.md` |
|
||||
| `references/improve.md` | Improve flow: signal verification, root-cause grouping, generalizing, delegation over growth, ADR-0020 retrofit |
|
||||
| `references/contract.md` | Description and body contract: three-part description shape, 250/400 tiers, delegation rule in place of a body word gate, invocation axis |
|
||||
| `references/plugin-scope.md` | Plugin/APM scope field rules for the single vendor-neutral file, plus its pre-audit checklist |
|
||||
| `references/project-user-scope.md` | Project/user scope field rules for the Claude Code + Copilot pair, both Copilot formats, plus its pre-audit checklist |
|
||||
| `references/deployment-modes.md` | Scope hierarchy and precedence, scoped identifiers, cache isolation, path conventions |
|
||||
| `references/scripts.md` | Conventions for new-agent.sh and the templates it copies: contract, template variables, file placement, error messages |
|
||||
| `references/sources.md` | Research provenance — sources that informed this skill |
|
||||
| `assets/templates/claude-code.md` | Annotated Claude Code agent definition template (project/user scope) |
|
||||
| `assets/templates/copilot.agent.md.template` | Annotated Copilot CLI agent definition template (project/user scope) |
|
||||
| `assets/templates/apm-agent.md` | Annotated vendor-neutral APM agent definition template (plugin/APM scope) |
|
||||
| `tests/new-agent.bats` | bats tests for `scripts/new-agent.sh` |
|
||||
| `tests/new-agent.bats` | (source-only) bats tests for `scripts/new-agent.sh` |
|
||||
| `assets/README.md` | Directory meta-documentation for assets/ |
|
||||
| `references/README.md` | Directory meta-documentation for references/ |
|
||||
| `scripts/README.md` | Directory meta-documentation for scripts/ |
|
||||
| `tests/README.md` | bats dependency instructions and run command |
|
||||
| `tests/README.md` | (source-only) bats dependency instructions and run command |
|
||||
|
||||
Rows marked **(source-only)** exist in the authoring source (`.apm/skills/agent-author/`) but are
|
||||
not present in an installed plugin: `scripts/sync-plugin-content.sh` strips
|
||||
`<category>/<name>/tests` when it generates the flat mirror, because these are dev-time fixtures no
|
||||
plugin host needs to discover (ADR-0017). Run them from a repo checkout, not from an install. The
|
||||
`assets/templates/` rows above are unaffected — the exclusion is depth-scoped to
|
||||
`<category>/<name>/tests`, so template trees that themselves contain a `tests/` directory ship
|
||||
intact.
|
||||
|
||||
@@ -1,18 +1,9 @@
|
||||
---
|
||||
name: agent-author
|
||||
description: >
|
||||
Use when the user wants to create a new agent definition file from scratch
|
||||
("write an agent for X", "build a subagent that does Y", "create an agent
|
||||
definition for Z"), or improve an existing one. Handles agent definitions at
|
||||
plugin/APM, project, and user scope. Project and user scope always generate
|
||||
a Claude Code (`.md`) + Copilot CLI (`.agent.md`) file pair in one pass;
|
||||
plugin/APM scope generates a single vendor-neutral `.apm/agents/<name>.agent.md`
|
||||
file instead (no per-target Claude Code / Copilot split). Also use when the
|
||||
user provides inline feedback about an agent's behavior and wants it applied,
|
||||
or when a grill session has produced findings the user wants acted on — even
|
||||
if they don't say "improve" explicitly. Do not use for read-only review —
|
||||
examine agent files manually or run a grill session to generate improvement
|
||||
signals. Do not use to author skills — use /skill-author instead.
|
||||
Use when the user wants to create a new agent definition file from scratch, or
|
||||
apply grill findings, audit findings, or inline feedback to an existing one.
|
||||
Not read-only review -> `agent-audit`. Not skills -> `skill-author`.
|
||||
allowed-tools: Bash Read Write Edit
|
||||
metadata:
|
||||
category: factory
|
||||
@@ -20,244 +11,54 @@ metadata:
|
||||
- context7-websites-code-claude
|
||||
- claude-code-plugins-docs
|
||||
- claude-code-subagents-docs
|
||||
- context7-github-en-copilot
|
||||
- github-custom-agents-configuration
|
||||
- github-cli-plugin-reference
|
||||
- github-plugins-creating
|
||||
---
|
||||
|
||||
## Gotchas
|
||||
|
||||
- At plugin/APM scope, bump the resolved package's `apm.yml` `version` after every change — minor for a new agent, patch for a fix. Consumers compare this version to detect updates; skipping it hides the change.
|
||||
- At plugin/APM scope, `tools` and all Claude-only fields (`isolation`, `maxTurns`, `effort`, `memory`, `permissionMode`, `disallowedTools`, `skills`, `color`, `initialPrompt`, `background`, `hooks`, `mcpServers`) are omitted entirely, not merely restricted (ADR-0016: `apm compile` copies frontmatter verbatim to both harnesses with no per-target integrator, so a harness-specific value is wrong on at least one). Only project/user scope supports these fields.
|
||||
- An `apm.yml` with no top-level `type:` field is a marketplace-only manifest, not a package root — the walk-up skips it and keeps going.
|
||||
- `AskUserQuestion`, `EnterPlanMode`, `ExitPlanMode`, `ScheduleWakeup`, and `WaitForMcpServers` are never available to any subagent regardless of the `tools` field. Exception: `ExitPlanMode` is available when the parent session runs in `permissionMode: plan`.
|
||||
- Duplicate `name` values in the same scope: Claude Code silently discards one without warning. Always verify uniqueness before shipping.
|
||||
- Plugin agents in subdirectories get scoped identifiers (`plugin:folder:name`) — keep agents flat in `agents/` to avoid this. Applies to project/user-scope Claude Code agents only.
|
||||
- Copilot CLI agent files **must** use the `.agent.md` extension — a plain `.md` file isn't picked up. The plugin/APM-scope single file also ends in `.agent.md` by convention, but it's vendor-neutral, not Copilot-only — it compiles to Claude Code too.
|
||||
- Copilot has no `permissionMode`, `maxTurns`, `isolation`, or `memory` fields — do not include them in project/user-scope Copilot files.
|
||||
- `model` resolution order for Claude Code: `CLAUDE_CODE_SUBAGENT_MODEL` env var → per-invocation parameter → frontmatter `model` → main session model. The frontmatter value is a low-priority default, not a guarantee.
|
||||
- At plugin/APM scope `tools` and every Claude-only field are omitted entirely, not merely ignored: `apm compile` copies frontmatter verbatim to both harnesses, so fencing a read-only agent with `tools:` is wrong on one of them. `disallowedTools` is the one restriction that survives (ADR-0016).
|
||||
- That fence is partial. It denies only the tools it names, never `Bash`, which a plugin-scope agent inherits — a shell redirect still writes. State the read-only boundary in the body too.
|
||||
- An agent body carries no word gate; delegation replaces it. A plugin/APM agent is one file with no sibling `references/` directory, so it cannot disclose to itself, only invoke skills — and a body restating a procedure an invocable skill owns is an `agent-audit` FAIL.
|
||||
- Duplicate `name` values in one scope: Claude Code discards one silently. Verify uniqueness before shipping.
|
||||
|
||||
## Route
|
||||
## Step 1 — Dispatch
|
||||
|
||||
If the destination resolves to plugin/APM scope (scope detection in Step 1 finds a `type:`-bearing `apm.yml` at or above the root), read `references/deployment-modes.md`.
|
||||
| Condition | Flow | Reference |
|
||||
|---|---|---|
|
||||
| No agent file at the target path(s) | Create | `references/create.md` |
|
||||
| A file exists, at least one improvement signal present | Improve | `references/improve.md` |
|
||||
| A file exists, no signals | Stop and ask | — |
|
||||
|
||||
Determine which flow before touching the filesystem:
|
||||
Signals: grill output, `agent-audit` findings, inline feedback, session context describing what went wrong. With none, ask: "No improvement signals found. Did you mean to create a new agent, or do you have feedback to apply?"
|
||||
|
||||
- **Neither `<name>.md` nor `<name>.agent.md` exist at the target paths** → follow **Creating a new agent**
|
||||
- **At least one file exists + improvement signals present** → follow **Improving an existing agent**
|
||||
- **At least one file exists + no signals** → ask: "No improvement signals found. Did you mean to create a new agent, or do you have feedback to apply?"
|
||||
Read only the reference for the resolved flow. Capture `git log --oneline -1` before touching the filesystem; Step 4 needs it.
|
||||
|
||||
Signals: grill session output, inline user feedback, session context describing what went wrong.
|
||||
## Step 2 — Scope
|
||||
|
||||
## Creating a new agent
|
||||
Scope decides which fields exist, so resolve it first. `scripts/new-agent.sh` walks up for a `type:`-bearing `apm.yml` and prints the scope it chose — read that output.
|
||||
|
||||
### Prerequisites
|
||||
| Resolved scope | Emits | Read |
|
||||
|---|---|---|
|
||||
| plugin/APM | one vendor-neutral `.apm/agents/<name>.agent.md` | `references/plugin-scope.md` |
|
||||
| project or user | a Claude Code `.md` + Copilot `.agent.md` pair | `references/project-user-scope.md` |
|
||||
|
||||
Before touching the filesystem, confirm you have:
|
||||
- [ ] Agent name (kebab-case, e.g. `code-reviewer`)
|
||||
- [ ] Root directory (a path inside a package for plugin/APM scope, project root, or `~` for user scope)
|
||||
- [ ] Agent purpose — one sentence describing the task this agent handles
|
||||
- [ ] Trigger condition — when should the runtime delegate to this agent?
|
||||
Read only the file for the resolved scope; the other describes fields this run cannot use. If precedence, cache isolation or path conventions matter, read `references/deployment-modes.md`.
|
||||
|
||||
If any are missing, stop and ask before proceeding. Then capture `git log --oneline -1` before touching the filesystem — Step 5 needs it to verify a real commit landed.
|
||||
## Step 3 — Contract
|
||||
|
||||
Verify `kyberforge:agent-audit` is available — it ships with the kyberforge plugin and is co-installed with this skill. If unavailable, stop and tell the user to install the kyberforge plugin before continuing.
|
||||
Before writing or editing a `description`, or restructuring a body, read `references/contract.md` — the three-part shape, banned content, the delegation rule and the body pattern.
|
||||
|
||||
### Step 1 — Scaffold
|
||||
Gates `agent-audit` enforces at every scope:
|
||||
|
||||
Run the scaffold script with the agent name and root directory:
|
||||
- **Description** — a trigger clause, at most one capability clause, and a boundary clause shaped `Not <thing> -> <name>` that resolves to a real skill or agent. 250 characters SUGGESTION, 400 FAIL, value only: an agent's `name` and `description` is preloaded into every session exactly as a skill's is.
|
||||
- **Body** — no word gate, and a delegation check in its place: name the skill to invoke rather than restating what it does.
|
||||
- **Invocation** — decide whether the agent is model-delegated or reached only by name. Only Copilot's cloud/IDE format expresses that in frontmatter (`disable-model-invocation`, `user-invocable`).
|
||||
|
||||
```bash
|
||||
bash scripts/new-agent.sh <name> <root>
|
||||
```
|
||||
At every scope, five tools reach no subagent whatever `tools` says — `AskUserQuestion`, `EnterPlanMode`, `ExitPlanMode`, `ScheduleWakeup`, `WaitForMcpServers`. Never write a body that has the agent ask the user a question or enter plan mode; it describes a turn the runtime cannot give it.
|
||||
|
||||
Examples:
|
||||
```bash
|
||||
bash scripts/new-agent.sh code-reviewer packages/my-package/ # plugin/APM scope if packages/my-package/apm.yml has a type: field
|
||||
bash scripts/new-agent.sh deploy-assistant .
|
||||
bash scripts/new-agent.sh security-reviewer ~
|
||||
```
|
||||
## Step 4 — Validate and close
|
||||
|
||||
**Scope detection (script handles this automatically).** The script walks up from `<root>` for a package boundary — same shape `agent-audit`'s `validate.sh` uses:
|
||||
- Nearest ancestor `apm.yml` with a top-level `type:` field (`instructions`/`skill`/`hybrid`/`prompts`) → **plugin/APM scope** → `<package-root>/.apm/agents/<name>.agent.md` (single vendor-neutral file). A `type:`-less `apm.yml` is marketplace-only — skipped, walk continues upward.
|
||||
- No such `apm.yml`, `<root>` is a project directory → **project scope** (unchanged) → `<root>/.claude/agents/<name>.md` + `<root>/.github/agents/<name>.agent.md`
|
||||
- `<root>` is exactly `~` (checked directly, no walk-up) → **user scope** (unchanged) → `~/.claude/agents/<name>.md` + `~/.copilot/agents/<name>.agent.md`
|
||||
Invoke `agent-audit` on each file written and resolve every FAIL before reporting done. It checks the field allowlist, name-to-stem match, leftover placeholders and template comments, the description budget and the Copilot body limit — do not hand-check those.
|
||||
|
||||
A bare `plugin.json` with no `apm.yml` no longer signals plugin scope — that path is fully replaced, not dual-mode; it falls through to project scope.
|
||||
At plugin/APM scope bump the resolved package's `apm.yml` `version` — **minor** on create, **patch** on improve — because consumers compare it to detect updates. Project and user scope have no manifest.
|
||||
|
||||
The script is file-by-file no-op — it skips any file that already exists.
|
||||
|
||||
### Step 2 — Fill in the agent file(s)
|
||||
|
||||
**At plugin/APM scope**, there is exactly one file: `<package-root>/.apm/agents/<name>.agent.md`. Frontmatter carries ONLY `name`, `description`, optionally `model`, and optionally `source_keys` (provenance metadata, not a runtime field — see the template) — never `tools` or the other Claude-only fields listed in Gotchas (ADR-0016). Fill in `name`, `description`, `model`, and the system prompt body per the guidance below; the rest of this step's field-by-field guidance (tools, maxTurns, effort, memory, isolation, disallowedTools, skills, color, initialPrompt, background) is project/user scope only. Skip Step 3 and go to Step 4.
|
||||
|
||||
**At project/user scope**, continue below to fill in both provider files — this step covers the Claude Code file (`<name>.md`); Step 3 covers the Copilot file.
|
||||
|
||||
Open the scaffolded Claude Code file. Replace every `FILL IN:` placeholder. **Remove all template documentation comments from the YAML frontmatter after filling in required fields** — these are marked with `<!--` and `-->` and must be deleted before shipping.
|
||||
|
||||
**`name`** — lowercase letters and hyphens only. Must be unique within the scope.
|
||||
|
||||
**`description`** — the most important field for autonomous delegation:
|
||||
- Start with an action verb: "Reviews...", "Analyzes...", "Generates..."
|
||||
- If this agent should trigger without explicit user direction, include "Use proactively" in the description
|
||||
- Specific about the triggering condition and expertise domain
|
||||
- Under 300 characters preferred
|
||||
|
||||
**`tools`** (project/user scope only — never at plugin/APM scope) — restrict to what the agent actually needs. Omit to inherit all tools. Use `Agent(type1,type2)` to limit which subagent types this agent can spawn; omit `Agent` entirely to prevent spawning.
|
||||
|
||||
**Optional fields worth considering (project/user scope only — never at plugin/APM scope):**
|
||||
- `model`: set when this agent needs a different capability tier (`haiku` for fast tasks, `opus` for deep reasoning)
|
||||
- `maxTurns`: set a cap to prevent runaway agents on bounded tasks
|
||||
- `effort`: set to `low` for single-lookup tasks, `high` or above for deep reasoning or multi-file analysis — overrides session effort level; omit to inherit
|
||||
- `memory`: `user`, `project`, or `local` — only when cross-session state is genuinely needed
|
||||
- `isolation: worktree` — only when the agent modifies files and needs an isolated copy
|
||||
- `disallowedTools`: space-separated denylist applied before `tools`; supports `mcp__*` glob patterns (e.g. `disallowedTools: mcp__filesystem__*`)
|
||||
- `skills`: list of skill names preloaded at agent startup — different from the `source_keys` metadata field
|
||||
- `color`: UI color for the agent tile (`red`, `blue`, `green`, `yellow`, `purple`, `orange`, `pink`, `cyan`)
|
||||
- `initialPrompt`: auto-submitted as the first turn when this agent activates as the main session thread; only set when this agent is intended for main-thread activation
|
||||
- `background`: set `true` to force background execution
|
||||
|
||||
**`source_keys`** — top-level list of research source slugs that informed this agent. Add only when research sources were used (i.e. entries with `` `extracted` `` status are in context from a prior `/research` session). Each slug must match an H2 heading in `sources.md` — see Step 4 for where that file lives (plugin/APM scope only). Omit entirely when no research was used.
|
||||
|
||||
```yaml
|
||||
source_keys:
|
||||
- my-source-slug
|
||||
```
|
||||
|
||||
**System prompt body** — write as a direct role instruction:
|
||||
- Open with: "You are a [role]. When invoked, [primary action]."
|
||||
- Cover: inputs expected, process steps, output format, error handling
|
||||
- One job per agent
|
||||
|
||||
### Step 3 — Fill in the Copilot agent file (project/user scope only)
|
||||
|
||||
Skip this step entirely at plugin/APM scope — there is no separate Copilot file there. The single `.apm/agents/<name>.agent.md` file from Step 2 already compiles to both Claude Code and Copilot CLI via `apm compile`.
|
||||
|
||||
**Two distinct Copilot agent formats** exist, with different paths and field sets. Choose one based on the deployment target:
|
||||
|
||||
**CLI format** (default — what the scaffold creates):
|
||||
- Path: `.github/agents/<name>.agent.md` (project) or `~/.copilot/agents/<name>.agent.md` (user)
|
||||
- Extension: **must be `.agent.md`**
|
||||
- Supported fields: `name` (required), `description` (required), `tools` (optional)
|
||||
- `tools` uses Copilot aliases: `execute` (shell), `read`, `edit`, `search`, `agent`, `web`
|
||||
- Body length limit: **30,000 characters** — content beyond this is silently truncated
|
||||
|
||||
**Cloud/IDE format** (use when targeting Copilot Chat in VS Code or GitHub.com):
|
||||
- Path: `.github/copilot/agents/<name>.md` (note: plain `.md`, different directory)
|
||||
- Additional fields available: `target` (`vscode`, `github-copilot`, or omit for both), `user-invocable` (set `false` to hide from manual invocation), `disable-model-invocation` (set `true` to require explicit user invocation), `mcp-servers` (MCP server config — processed by cloud runtime, ignored in VS Code)
|
||||
- Body length limit: **30,000 characters** — silently truncated
|
||||
|
||||
**Do not include Claude Code-only fields in either format**: `maxTurns`, `isolation`, `memory`, `permissionMode`, `effort`, `hooks`, `mcpServers`.
|
||||
|
||||
**`source_keys`** — add the same top-level list as the CC file when research sources were used. Omit when no research was used.
|
||||
|
||||
**Remove all template documentation comments from the YAML frontmatter after filling in required fields** — these are marked with `<!--` and `-->` and must be deleted before shipping.
|
||||
|
||||
The system prompt body should match the Claude Code version — the agent's task definition is the same across providers.
|
||||
|
||||
### Step 4 — Populate or delete `sources.md` (plugin/APM scope only)
|
||||
|
||||
Skip at project/user scope. The file lives at the package root (alongside `apm.yml`), not inside `.apm/agents/` — otherwise tooling that scans that directory for agent definitions would treat it as an agent needing frontmatter (ADR-0010).
|
||||
|
||||
If a research `sources.md` is present in the conversation context:
|
||||
1. Filter to entries with `` `extracted` `` status only.
|
||||
2. For each entry, identify which agent file it contributed to.
|
||||
3. Write `sources.md` at the package root using the format below. Paths in `Contributing files:` are relative to the package root.
|
||||
|
||||
```markdown
|
||||
# Sources
|
||||
|
||||
## slug-name
|
||||
|
||||
- **URL:** <source URL>
|
||||
- **Research doc:** <path/to/research/sources.md relative to repo root>
|
||||
- **Description:** <what this source covers>
|
||||
- **Contributing files:** .apm/agents/<name>.agent.md
|
||||
- **Status:** `extracted`
|
||||
```
|
||||
|
||||
Each slug must match an H2 heading, and each slug must also appear in the `source_keys` list of the file listed under `Contributing files:`.
|
||||
|
||||
If no research sources are in context, delete `sources.md`.
|
||||
|
||||
### Step 5 — Validate and close
|
||||
|
||||
Run this checklist before invoking the audit:
|
||||
|
||||
**Plugin/APM scope — single file (`<name>.agent.md`):**
|
||||
- [ ] `name` field present, kebab-case, unique in scope
|
||||
- [ ] `description` field present and action-first
|
||||
- [ ] Frontmatter contains ONLY `name`, `description`, and optionally `model` (plus `source_keys` if research-sourced) — no `tools`, `isolation`, `maxTurns`, `effort`, `memory`, `permissionMode`, `disallowedTools`, `skills`, `color`, `initialPrompt`, `background`, `hooks`, or `mcpServers`
|
||||
- [ ] System prompt body present and non-empty
|
||||
- [ ] No `FILL IN:` placeholders remain
|
||||
- [ ] No `<!-- -->` template comments remain in frontmatter
|
||||
|
||||
**Project/user scope — Claude Code file (`<name>.md`):**
|
||||
- [ ] `name` field present, kebab-case, unique in scope
|
||||
- [ ] `description` field present and action-first
|
||||
- [ ] System prompt body present and non-empty
|
||||
- [ ] No `FILL IN:` placeholders remain
|
||||
- [ ] No `<!-- -->` template comments remain in frontmatter
|
||||
|
||||
**Project/user scope — Copilot CLI file (`<name>.agent.md`):**
|
||||
- [ ] File extension is `.agent.md` (not `.md`)
|
||||
- [ ] `name` field matches the filename stem (e.g. `name: my-agent` in `my-agent.agent.md`)
|
||||
- [ ] `description` field present
|
||||
- [ ] No Claude Code-only fields (`maxTurns`, `isolation`, `memory`, `permissionMode`, `effort`, `hooks`, `mcpServers`)
|
||||
- [ ] System prompt body present and non-empty
|
||||
- [ ] Body does not exceed 30,000 characters
|
||||
- [ ] No `<!-- -->` template comments remain in frontmatter
|
||||
|
||||
At plugin/APM scope, apply a **minor bump** to the resolved package's `apm.yml` `version` (single manifest, e.g. `1.0.4` → `1.1.0`).
|
||||
|
||||
Invoke `kyberforge:agent-audit` on the created file(s) before closing — validates the pair at project/user scope, the single file at plugin/APM scope.
|
||||
|
||||
**Commit verification.** Capture `git log --oneline -1` before Step 1 and keep it. Once the audit is clean, run `git add` and `git commit` for the new agent files — do not stop at staging. Then run `git log --oneline -1` again and confirm the hash changed from the one you captured at the start. A non-empty `git diff --stat` is not sufficient proof of completion: staged-but-uncommitted work isn't part of any commit and can be silently lost if the working tree is cleaned up before a commit lands. Only report the agent as done once the hash has actually changed.
|
||||
|
||||
## Improving an existing agent
|
||||
|
||||
### Step 1 — Verify inputs
|
||||
|
||||
Confirm the agent files exist and at least one improvement signal is present in the conversation or a referenced file.
|
||||
|
||||
If no signals: "This skill applies existing signals to an agent. For a blind review, examine the files manually or run a grill session first."
|
||||
|
||||
Verify `kyberforge:agent-audit` is available — it ships with the kyberforge plugin and is co-installed with this skill. If unavailable, stop and tell the user to install the kyberforge plugin before continuing.
|
||||
|
||||
Capture `git log --oneline -1` now, before making any edits — Step 5 needs it to verify a real commit landed.
|
||||
|
||||
**Partial state (project/user scope only)** — if one provider file exists but not the other, scaffold the missing one (`bash scripts/new-agent.sh <name> <root>`, file-by-file no-op) then continue. Doesn't apply at plugin/APM scope — single file, no partial-pair state.
|
||||
|
||||
### Step 2 — Gather and group signals
|
||||
|
||||
Read the current agent file(s). Collect all signals from the conversation.
|
||||
|
||||
Group by **root cause**, not symptom. One root cause → one fix.
|
||||
|
||||
```text
|
||||
Example:
|
||||
- User feedback: agent keeps trying to push to remote
|
||||
- Session context: no scope boundary in system prompt
|
||||
→ Root cause: system prompt lacks git scope constraint → fix: add explicit boundary
|
||||
```
|
||||
|
||||
### Step 3 — Announce planned changes
|
||||
|
||||
Before editing, state which root causes were identified, what evidence supports each, and which files will change. Then proceed — edits are reversible via git.
|
||||
|
||||
### Step 4 — Apply changes
|
||||
|
||||
Edit any file the signals point to. Generalize the fix — find the underlying gap, not the specific example that failed. For every sentence you add, ask: "Would the agent get this wrong without it?" A shorter, focused definition consistently outperforms an exhaustive one. For Copilot files, verify no Claude Code-only fields are introduced. For a plugin/APM-scope single file, verify no field beyond `name`, `description`, `model`, and `source_keys` is introduced.
|
||||
|
||||
If the edit adds or removes research-sourced content, update `source_keys` in the edited file(s) and the corresponding entry in `sources.md` per Create flow's Step 4.
|
||||
|
||||
### Step 5 — Validate and close
|
||||
|
||||
Re-run the validation checklist from the create flow's Step 5 on any edited file.
|
||||
|
||||
At plugin/APM scope, apply a **patch bump** to the resolved package's `apm.yml` `version` (e.g. `1.0.4` → `1.0.5`).
|
||||
|
||||
Invoke `kyberforge:agent-audit` on the edited file(s) to confirm no regressions — the pair at project/user scope, the single file at plugin/APM scope.
|
||||
|
||||
**Commit verification.** Capture `git log --oneline -1` at the start of Step 1 and keep it. Once the audit is clean, run `git add` and `git commit` for the changed files — do not stop at staging. Then run `git log --oneline -1` again and confirm the hash changed from the one you captured at the start. A non-empty `git diff --stat` is not sufficient proof of completion: staged-but-uncommitted work isn't part of any commit and can be silently lost if the working tree is cleaned up before a commit lands. Only report the improvement as done once the hash has actually changed.
|
||||
**Commit verification.** Once the audit is clean, run `git add` and `git commit` — do not stop at staging. Re-run `git log --oneline -1` and confirm the hash changed from Step 1's. A non-empty `git diff --stat` is not proof: staged-but-uncommitted work is part of no commit and is lost if the tree is cleaned up. Report done only once the hash has changed.
|
||||
|
||||
@@ -3,7 +3,8 @@
|
||||
## templates/
|
||||
|
||||
Annotated agent definition templates copied by `scripts/new-agent.sh` when scaffolding a new agent.
|
||||
All three scaffold the `description` in the three-part ADR-0020 shape — a `Use when` trigger clause, at most one capability clause, and a boundary clause — rather than the deleted action-verb opener, and each carries a delegate-don't-restate note in the body.
|
||||
|
||||
- **`claude-code.md`** — Claude Code agent definition template (project/user scope). Includes all supported frontmatter fields (required and optional) with inline guidance comments and `FILL IN:` placeholders.
|
||||
- **`copilot.agent.md.template`** — Copilot CLI agent definition template (CLI format, project/user scope). Excludes cloud/IDE-only fields (`target`, `user-invocable`, `disable-model-invocation`, `mcp-servers`) and Claude Code-only fields. Uses Copilot tool aliases (`execute`, `read`, `edit`, `search`, `agent`, `web`).
|
||||
- **`apm-agent.md`** — Vendor-neutral APM agent definition template (plugin/APM scope). Only `name`, `description`, optional `model`, and optional `source_keys` (provenance metadata, not a runtime field) in frontmatter — no `tools` and no Claude-only fields, since `apm compile` copies frontmatter verbatim to both the Claude Code and Copilot CLI targets with no per-target integrator (ADR-0016).
|
||||
- **`apm-agent.md`** — Vendor-neutral APM agent definition template (plugin/APM scope). Frontmatter is limited to the `apm-agent-allowlist` section of `agent-audit`'s `references/field-inventory.md` — the authoritative list, read from there as data by `agent-audit`'s `validate.sh`; this file deliberately does not restate it. No `tools` and no Claude-only knobs, since `apm compile` copies frontmatter verbatim to both the Claude Code and Copilot CLI targets with no per-target integrator; `disallowedTools` is scaffolded as an opt-in comment because a denylist, unlike the `tools` allowlist, survives that copy (ADR-0016 and its 2026-08-14 amendment).
|
||||
|
||||
@@ -2,34 +2,60 @@
|
||||
<!-- Vendor-neutral APM agent definition (plugin/APM scope).
|
||||
Path: <package-root>/.apm/agents/<name>.agent.md — one file, no counterpart.
|
||||
`apm compile` copies this frontmatter verbatim to BOTH the Claude Code and
|
||||
Copilot CLI targets — there is no per-target field integrator. Claude's
|
||||
`tools:` (space-separated string) and Copilot's `tools:` (alias list) are
|
||||
incompatible vocabularies, and Claude-only fields (isolation, maxTurns,
|
||||
effort, memory, permissionMode) have no Copilot equivalent. A value correct
|
||||
for one harness is guaranteed wrong on the other, so this scope carries
|
||||
ONLY the fields below — full stop (see ADR-0016). `source_keys` is
|
||||
provenance metadata, not a runtime field, and is exempt from that rule.
|
||||
Copilot CLI targets, with no per-target field integrator to reconcile
|
||||
anything, so a harness-specific value is wrong on at least one of them.
|
||||
|
||||
Do NOT add: tools, isolation, maxTurns, effort, memory, permissionMode,
|
||||
disallowedTools, skills, color, initialPrompt, background, hooks, or
|
||||
mcpServers. Omitting `tools` means inherit-all-tools on both harnesses,
|
||||
which is never wrong.
|
||||
This template does not restate the permitted-field list. The authoritative
|
||||
list is the `apm-agent-allowlist` section of agent-audit's
|
||||
references/field-inventory.md, which agent-audit's validate.sh reads from
|
||||
there as data — a list copied into a template goes stale one step further
|
||||
out than the list itself. Every field scaffolded below is on it; before
|
||||
adding any other field, check that section.
|
||||
|
||||
The shape rule behind the list (ADR-0016 and its 2026-08-14 amendment):
|
||||
`tools` is an ALLOWLIST whose vocabulary differs per harness — Claude tool
|
||||
names vs Copilot's execute/read/edit/search/agent/web — so one value is
|
||||
wrong on one target. Never add it here; omitting it means inherit-all-tools
|
||||
on both harnesses, which is never wrong. `disallowedTools` is a DENYLIST
|
||||
and is allowed for exactly that reason: a name the other harness does not
|
||||
recognise denies nothing, so the worst case is a missing fence, never a
|
||||
wrongly granted capability. Claude-only knobs (isolation, maxTurns, effort,
|
||||
memory, permissionMode) have no Copilot equivalent and stay out.
|
||||
|
||||
Fill in all FILL IN: placeholders. Delete template comments before shipping. -->
|
||||
|
||||
name: AGENT_NAME
|
||||
<!-- Required. Lowercase letters and hyphens only. Must be unique within the scope. -->
|
||||
|
||||
description: FILL IN: Action-first description of what this agent does and when to invoke it.
|
||||
<!-- Required. The primary signal for autonomous delegation.
|
||||
Start with a verb: "Reviews...", "Analyzes...", "Generates..."
|
||||
Be specific about the triggering condition and expertise domain.
|
||||
Example: "Reviews pull request diffs for security issues. Use proactively after code changes." -->
|
||||
description: FILL IN: Use when <trigger>. <One capability clause.> Not <thing> -> <name>.
|
||||
<!-- Required. The primary signal for autonomous delegation, and preloaded into every
|
||||
session whether or not this agent is ever used. Three parts, nothing else:
|
||||
a trigger clause opening "Use when", at most one capability clause, and a
|
||||
boundary clause naming a real sibling skill or agent.
|
||||
250 characters is the target, 400 the hard ceiling (ADR-0020).
|
||||
Do not open with an action verb ("Reviews...", "Analyzes...") — that rule was
|
||||
deleted.
|
||||
Never write "Use proactively" here. It steers the Claude Code runtime and does
|
||||
nothing anywhere else, and this file compiles to a Copilot `.agent.md` too, where
|
||||
agent-audit's KyberforgeCopilot.ProactivePhrase rule grades it a hard FAIL.
|
||||
The phrase is CC-only; at this scope, a precise trigger clause does that job.
|
||||
Example: "Use when a diff needs checking for injected credentials before it
|
||||
merges. Not prose or style linting -> `lint-runner`." -->
|
||||
|
||||
<!-- model: sonnet
|
||||
Optional. Aliases: sonnet, opus, haiku, fable. Or full model ID.
|
||||
Omit to inherit the runtime default on whichever harness compiles this file. -->
|
||||
|
||||
<!-- disallowedTools: Edit, Write, NotebookEdit
|
||||
Optional. Denylist, applied before `tools` and taking precedence over it.
|
||||
Add it when this agent is read-only — it is the one tool restriction that
|
||||
survives verbatim copy (see the header comment). Claude Code honours it for
|
||||
plugin subagents — confirmed. Copilot's handling of the key is unconfirmed;
|
||||
ADR-0016 accepts that as a stated risk rather than a settled fact.
|
||||
It denies only the tools it names. It does NOT deny Bash, which this agent
|
||||
inherits, so a shell redirect still writes — state the read-only boundary
|
||||
in the system prompt body as well, not in frontmatter alone. -->
|
||||
|
||||
<!-- source_keys:
|
||||
- slug-name
|
||||
Development-only. Add when research sources informed this agent (slugs must match
|
||||
@@ -41,6 +67,11 @@ FILL IN: System prompt body. Write as a direct role instruction.
|
||||
|
||||
You are a FILL IN: role description. When invoked, FILL IN: primary action.
|
||||
|
||||
<!-- Delegate, don't restate. If an installed skill already owns a procedure this agent
|
||||
needs, name it ("invoke `git-commits`") instead of transcribing it — a body that
|
||||
restates a procedure an invocable skill owns is an agent-audit FAIL. One job per
|
||||
agent. Delete this comment before shipping. -->
|
||||
|
||||
## Inputs
|
||||
|
||||
FILL IN: What inputs does this agent expect? (files, context, parameters)
|
||||
@@ -52,3 +83,10 @@ FILL IN: Steps the agent takes. Be specific about ordering if it matters.
|
||||
## Output
|
||||
|
||||
FILL IN: What does the agent produce? Format, location, structure.
|
||||
|
||||
## Errors
|
||||
|
||||
FILL IN: What does the agent do on malformed, missing or contradictory input?
|
||||
State whether it stops and reports, or degrades to a named fallback — and what it
|
||||
tells the caller either way. An agent with no error handling invents a recovery,
|
||||
and an invented recovery is invisible until the output is wrong.
|
||||
|
||||
@@ -7,20 +7,31 @@ name: AGENT_NAME
|
||||
<!-- Required. Lowercase letters and hyphens only. Must be unique within the scope.
|
||||
Duplicate names are silently discarded — no warning is emitted. -->
|
||||
|
||||
description: FILL IN: Action-first description of what this agent does and when to invoke it.
|
||||
<!-- Required. The primary signal for autonomous delegation.
|
||||
Start with a verb: "Reviews...", "Analyzes...", "Generates..."
|
||||
Include "Use proactively" to trigger automatic invocation.
|
||||
Be specific about the triggering condition and domain.
|
||||
Example: "Reviews pull request diffs for security issues. Use proactively after code changes." -->
|
||||
description: FILL IN: Use when <trigger>. <One capability clause.> Not <thing> -> <name>.
|
||||
<!-- Required. The primary signal for autonomous delegation, and preloaded into every
|
||||
session whether or not this agent is ever used. Three parts, nothing else:
|
||||
a trigger clause opening "Use when", at most one capability clause, and a
|
||||
boundary clause naming a real sibling skill or agent.
|
||||
250 characters is the target, 400 the hard ceiling (ADR-0020).
|
||||
Do not open with an action verb ("Reviews...", "Analyzes...") — that rule was
|
||||
deleted.
|
||||
"Use proactively" is valid HERE and only here: it steers the Claude Code runtime
|
||||
to offer this agent unprompted. Add it only if that is what you want. If you add
|
||||
it, leave it OUT of the Copilot half of the pair — the phrase does nothing there
|
||||
and agent-audit's KyberforgeCopilot.ProactivePhrase grades it a hard FAIL. The
|
||||
pair must describe the same job; it does not have to be byte-identical.
|
||||
Example: "Use when a diff needs checking for injected credentials before it
|
||||
merges. Not prose or style linting -> `lint-runner`." -->
|
||||
|
||||
<!-- tools: Read Bash Grep
|
||||
Optional. Space-separated allowlist. Omit to inherit all tools from parent.
|
||||
<!-- tools: Read, Bash, Grep
|
||||
Optional. Allowlist of tool names: a comma-separated string or a YAML list.
|
||||
Restrict it to what the agent actually needs. Omit only when it needs them
|
||||
all — omitting inherits every tool from the parent.
|
||||
Use Agent(type1,type2) to restrict which subagent types this agent can spawn.
|
||||
Omit Agent entirely to prevent this agent from spawning subagents.
|
||||
Never available to subagents regardless of tools field:
|
||||
AskUserQuestion, EnterPlanMode, ExitPlanMode, ScheduleWakeup, WaitForMcpServers
|
||||
Exception: ExitPlanMode IS available when parent session runs in permissionMode: plan -->
|
||||
Listing any of them is a finding: agent-audit enforces the flat rule. -->
|
||||
|
||||
<!-- model: sonnet
|
||||
Optional. Aliases: sonnet, opus, haiku, fable. Or full model ID.
|
||||
@@ -47,9 +58,13 @@ description: FILL IN: Action-first description of what this agent does and when
|
||||
<!-- background: false
|
||||
Optional. Set true to force background execution. -->
|
||||
|
||||
<!-- disallowedTools: mcp__filesystem__write_file
|
||||
Optional. Space-separated denylist, applied before the tools allowlist.
|
||||
Supports mcp__* glob patterns (e.g. mcp__filesystem__* to block all filesystem tools). -->
|
||||
<!-- disallowedTools: Edit, Write, NotebookEdit
|
||||
Optional. Denylist, applied before the tools allowlist and taking precedence over it.
|
||||
Accepts a YAML list or a delimited string; use the comma-separated string form for
|
||||
consistency with the plugin-scope agents in this repo.
|
||||
Supports mcp__* glob patterns (e.g. mcp__filesystem__* to block all filesystem tools).
|
||||
Denies only the tools it names — it does not deny Bash, so an agent that inherits
|
||||
Bash can still write via a shell redirect. State read-only intent in the body too. -->
|
||||
|
||||
<!-- skills:
|
||||
- skill-name
|
||||
@@ -73,6 +88,11 @@ FILL IN: System prompt body. Write as a direct role instruction.
|
||||
|
||||
You are a FILL IN: role description. When invoked, FILL IN: primary action.
|
||||
|
||||
<!-- Delegate, don't restate. If an installed skill already owns a procedure this agent
|
||||
needs, name it ("invoke `git-commits`") instead of transcribing it — a body that
|
||||
restates a procedure an invocable skill owns is an agent-audit FAIL. One job per
|
||||
agent. Delete this comment before shipping. -->
|
||||
|
||||
## Inputs
|
||||
|
||||
FILL IN: What inputs does this agent expect? (files, context, parameters)
|
||||
@@ -84,3 +104,10 @@ FILL IN: Steps the agent takes. Be specific about ordering if it matters.
|
||||
## Output
|
||||
|
||||
FILL IN: What does the agent produce? Format, location, structure.
|
||||
|
||||
## Errors
|
||||
|
||||
FILL IN: What does the agent do on malformed, missing or contradictory input?
|
||||
State whether it stops and reports, or degrades to a named fallback — and what it
|
||||
tells the caller either way. An agent with no error handling invents a recovery,
|
||||
and an invented recovery is invisible until the output is wrong.
|
||||
|
||||
@@ -12,10 +12,20 @@
|
||||
name: AGENT_NAME
|
||||
<!-- Required. Kebab-case identifier. Home-directory version wins on name collision. -->
|
||||
|
||||
description: FILL IN: Action-first description of what this agent does and when to invoke it.
|
||||
<!-- Required. Used by the runtime for automatic agent selection — quality matters.
|
||||
Start with a verb: "Reviews...", "Analyzes...", "Generates..."
|
||||
Example: "Reviews pull request diffs for security issues." -->
|
||||
description: FILL IN: Use when <trigger>. <One capability clause.> Not <thing> -> <name>.
|
||||
<!-- Required. Used by the runtime for automatic agent selection, and preloaded into
|
||||
every session whether or not this agent is ever used. Three parts, nothing else:
|
||||
a trigger clause opening "Use when", at most one capability clause, and a
|
||||
boundary clause naming a real sibling skill or agent.
|
||||
250 characters is the target, 400 the hard ceiling (ADR-0020).
|
||||
Do not open with an action verb ("Reviews...", "Analyzes...") — that rule was deleted.
|
||||
Never write "Use proactively" here. It steers the Claude Code runtime and does nothing
|
||||
in Copilot, and agent-audit's KyberforgeCopilot.ProactivePhrase grades it a hard FAIL.
|
||||
Otherwise keep the wording matched to the Claude Code half of the pair: agent-audit
|
||||
checks that both halves describe the same job, not that they are byte-identical, so
|
||||
dropping the CC-only phrase here is not a pair-consistency finding.
|
||||
Example: "Use when a diff needs checking for injected credentials before it
|
||||
merges. Not prose or style linting -> `lint-runner`." -->
|
||||
|
||||
<!-- tools: ["read", "search", "edit"]
|
||||
Optional. Array of tool names. Omit = all available tools. [] = no tools.
|
||||
@@ -46,6 +56,11 @@ FILL IN: System prompt body. Should match the Claude Code version — the agent'
|
||||
|
||||
You are a FILL IN: role description. When invoked, FILL IN: primary action.
|
||||
|
||||
<!-- Delegate, don't restate. If an installed skill already owns a procedure this agent
|
||||
needs, name it ("invoke `git-commits`") instead of transcribing it — a body that
|
||||
restates a procedure an invocable skill owns is an agent-audit FAIL. One job per
|
||||
agent. Delete this comment before shipping. -->
|
||||
|
||||
## Inputs
|
||||
|
||||
FILL IN: What inputs does this agent expect? (files, context, parameters)
|
||||
@@ -57,3 +72,10 @@ FILL IN: Steps the agent takes. Be specific about ordering if it matters.
|
||||
## Output
|
||||
|
||||
FILL IN: What does the agent produce? Format, location, structure.
|
||||
|
||||
## Errors
|
||||
|
||||
FILL IN: What does the agent do on malformed, missing or contradictory input?
|
||||
State whether it stops and reports, or degrades to a named fallback — and what it
|
||||
tells the caller either way. An agent with no error handling invents a recovery,
|
||||
and an invented recovery is invisible until the output is wrong.
|
||||
|
||||
@@ -4,14 +4,52 @@ source_keys: []
|
||||
|
||||
# references/
|
||||
|
||||
## create.md
|
||||
|
||||
The create flow, loaded from SKILL.md Step 1 when no agent file exists at the target path.
|
||||
Covers: prerequisites, the scaffold script and its scope walk-up, what to fill in at every scope,
|
||||
and populating or deleting the package-root `sources.md`.
|
||||
|
||||
## improve.md
|
||||
|
||||
The improve flow, loaded from SKILL.md Step 1 when a file exists and at least one improvement
|
||||
signal is present. Covers: signal verification, partial-pair recovery, root-cause grouping,
|
||||
generalizing rather than patching, delegation over growth, and the ADR-0020 retrofit rule.
|
||||
|
||||
## contract.md
|
||||
|
||||
The description and body contract, loaded from SKILL.md Step 3 before any description is written
|
||||
or any body restructured. Covers: the three-part description shape, banned description content,
|
||||
boundary-target resolution, the 250/400 length tiers, the body role-instruction pattern, the
|
||||
delegation rule that replaces a body word gate, and the invocation axis.
|
||||
|
||||
## plugin-scope.md
|
||||
|
||||
Field rules and the pre-audit checklist for the single vendor-neutral `.apm/agents/<name>.agent.md`
|
||||
file. Loaded from SKILL.md Step 2 when the scaffold resolves plugin/APM scope.
|
||||
|
||||
## project-user-scope.md
|
||||
|
||||
Field rules and the pre-audit checklist for the Claude Code `.md` + Copilot `.agent.md` pair,
|
||||
including the two distinct Copilot formats. Loaded from SKILL.md Step 2 when the scaffold resolves
|
||||
project or user scope.
|
||||
|
||||
## deployment-modes.md
|
||||
|
||||
Agent scope hierarchy, precedence rules, and per-scope restrictions. Covers: which fields are silently ignored for plugin agents (Claude Code and Copilot CLI), scoped identifiers for plugin subdirectory agents, cache isolation behaviour, and Copilot CLI path conventions. Loaded conditionally from SKILL.md when the destination is a plugin directory.
|
||||
Scope hierarchy and precedence, scoped identifiers for plugin subdirectory agents, cache isolation
|
||||
behaviour, and Copilot CLI path conventions. Loaded from SKILL.md Step 2 when precedence, paths or
|
||||
cache isolation matter to the run.
|
||||
|
||||
## scripts.md
|
||||
|
||||
Conventions for the `new-agent.sh` scaffold script and any future scripts added to this skill. Covers: what scripts should and should not do, file placement, error handling, template variable conventions, and the no-interactive-prompts rule.
|
||||
Conventions for the `new-agent.sh` scaffold script, the templates it copies, and any future script
|
||||
in this skill. Loaded from `create.md` Step 1 when the script or a template has to change. Covers:
|
||||
the no-interactive-prompts rule, structured output, idempotency, template variables, file
|
||||
placement, error messages, and the no-restated-field-roster rule that `tests/new-agent.bats`
|
||||
enforces.
|
||||
|
||||
## sources.md
|
||||
|
||||
Research provenance record for this skill. Lists the upstream research sources (claude-code-plugins and github-copilot-plugins research docs) that informed SKILL.md, the templates, and the deployment-modes reference. Used by `skill-audit` to validate the provenance chain.
|
||||
Research provenance record for this skill. Lists the upstream research sources
|
||||
(claude-code-plugins and github-copilot-plugins research docs) that informed SKILL.md and the
|
||||
reference files. Used by `skill-audit` to validate the provenance chain.
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
---
|
||||
source_keys:
|
||||
- claude-code-subagents-docs
|
||||
- github-custom-agents-configuration
|
||||
---
|
||||
|
||||
# The agent description and body contract
|
||||
|
||||
House contract, set by ADR-0020. The counts and the boundary targets are enforced by
|
||||
`agent-audit`'s `scripts/validate.sh`; the prose patterns by the Vale styles it bundles; the
|
||||
judgment calls by its reference files.
|
||||
|
||||
## Why the budget exists
|
||||
|
||||
An agent's `name` and `description` is loaded into every session's context at startup, whether or
|
||||
not the agent is ever delegated to — the same cost a skill's description carries, so agents take
|
||||
the same numbers. The body is different: it is not loaded into the caller's conversation at all,
|
||||
it *becomes the system prompt of a fresh context* when the agent runs. That is why the body has no
|
||||
word gate here and a skill body has one.
|
||||
|
||||
## Description
|
||||
|
||||
A description carries exactly three things:
|
||||
|
||||
1. **Trigger clause** — when to delegate, imperative: "Use when …", never "This agent …". Describe
|
||||
the user's intent and the triggering condition, not the agent's internal mechanics.
|
||||
2. **At most one capability clause** — what it does, one clause, no enumeration. Be specific
|
||||
("reviews a diff for injected credentials", not "helps with security").
|
||||
3. **Boundary clause** — form: `Not <thing> -> <name>.` Add one only where a near-miss agent or
|
||||
skill could steal delegations.
|
||||
|
||||
Banned from a description; move it to the body or to `README.md`:
|
||||
|
||||
- Capability enumeration or feature lists
|
||||
- Per-scope emission mechanics — which files the author skill writes at which scope changes no
|
||||
delegation decision
|
||||
- Output-format detail ("Produces a compact findings report with Why and Fix per finding")
|
||||
- Composition or architecture notes ("composes X rather than duplicating Y", "cross-cutting")
|
||||
- Implementation detail ("Self-validates via a bundled deterministic script")
|
||||
- Restating the same trigger twice in two registers — a verb list, then the same verbs re-quoted
|
||||
as user phrasings. This is a FAIL, not a suggestion.
|
||||
|
||||
**Do not open with an action verb.** "Reviews…", "Analyzes…", "Generates…" was the old house rule
|
||||
and ADR-0020 deleted it: the opener is `Use when`, matching every skill in this corpus, so one
|
||||
router reads one shape.
|
||||
|
||||
**"Use proactively" is Claude Code-only, and conditional even there.** The phrase steers the
|
||||
Claude Code runtime to offer an agent unprompted and does nothing anywhere else, so where it may
|
||||
appear depends on the file:
|
||||
|
||||
| File | Rule |
|
||||
|---|---|
|
||||
| Claude Code `.md` (project/user scope) | Allowed. Add it only where the runtime should delegate without the user naming the agent — an agent invoked by name does not need it, and it costs activations elsewhere when added by reflex. |
|
||||
| Copilot `.agent.md` (project/user scope) | **Never.** Inert there, and `KyberforgeCopilot.ProactivePhrase` grades it a hard FAIL. |
|
||||
| Vendor-neutral `.apm/agents/<name>.agent.md` (plugin/APM scope) | **Never.** Same Vale rule, same hard FAIL — the file matches the `**/*.agent.md` glob, and it compiles to a real Copilot agent downstream. |
|
||||
|
||||
A pair whose Claude Code half carries the phrase and whose Copilot half omits it is correct, not
|
||||
inconsistent: `agent-audit` checks that both halves describe the same job, not that they match
|
||||
word for word.
|
||||
|
||||
Indirect triggers ("even if the user doesn't say X") take a similar conditional at every scope:
|
||||
add one only where the user's natural phrasing genuinely omits the domain word.
|
||||
|
||||
**Boundary targets must resolve.** Both forms are checked — the arrow and the prose form ("do not
|
||||
use for X, use `y` instead") — so a typo dangles either way. Targets resolve against a universe
|
||||
built by walking up **from the agent file itself**: the nearest ancestor holding
|
||||
`plugins/*/.apm/{skills,agents}` (or, failing that, the nearest ancestor holding `.git`) contributes
|
||||
every skill and agent under `<root>/plugins/*/`, plus the agent's own apm package and the packages
|
||||
that package declares in `apm.yml` under `dependencies.apm`. A sibling plugin in the same monorepo
|
||||
therefore resolves; a skill in an unrelated repo does not. A target outside that universe sends the
|
||||
router nowhere. Verify it before writing it — do not invent a plausible sibling.
|
||||
|
||||
That universe is the apm marketplace and stops there. A **host built-in is not a routing target**:
|
||||
`/compact`, `/clear` and `/init` are Claude Code slash commands with no counterpart in Copilot CLI
|
||||
or Codex, and `.apm/` source compiles for all three, so routing to one is a portability defect. The
|
||||
gate is right to fail it and there is no allowlist. If a built-in genuinely needs mentioning, write
|
||||
it un-slashed — ``the `compact` built-in`` — which makes no routing claim and is not checked.
|
||||
|
||||
**Length.** 250 characters SUGGESTION, 400 characters FAIL, counting the frontmatter value only
|
||||
with YAML folding resolved. Treat 250 as the target: the SUGGESTION tier is what moves the corpus
|
||||
average, the FAIL tier only stops outliers.
|
||||
|
||||
## Body
|
||||
|
||||
Write the body as a direct role instruction, addressed to the agent:
|
||||
|
||||
````markdown
|
||||
You are a <role>. When invoked, <primary action>.
|
||||
|
||||
## Inputs
|
||||
<what the agent is given: files, context, parameters>
|
||||
|
||||
## Process
|
||||
<ordered steps; be explicit where ordering matters>
|
||||
|
||||
## Output
|
||||
<what it produces: format, location, structure>
|
||||
|
||||
## Errors
|
||||
<what to do on malformed, missing or contradictory input: report and stop, or
|
||||
which fallback to take — and what to say to the caller either way>
|
||||
````
|
||||
|
||||
Four required elements: **inputs expected, process steps, output format, error handling.** The
|
||||
last is the one that gets dropped, and dropping it is not neutral: an agent given a malformed
|
||||
input and no instruction invents a recovery, and a subagent's invented recovery is invisible to
|
||||
the caller until the output is wrong. Say explicitly whether the agent stops and reports, or
|
||||
degrades to a named fallback.
|
||||
|
||||
One job per agent. An agent covering two jobs gets delegated to for the wrong one.
|
||||
|
||||
**Delegation discipline replaces the word gate.** A plugin/APM agent is a single file with no
|
||||
sibling `references/` directory: it cannot disclose progressively to itself, so its only way to
|
||||
stay short is to *invoke* rather than *restate*. A body that transcribes a procedure a skill it
|
||||
can invoke already owns is an `agent-audit` FAIL, and the fix is one line — "invoke `<skill>`".
|
||||
|
||||
- Restating: "To commit, check the message against Conventional Commits: type, scope,
|
||||
description; header under 100 chars; …"
|
||||
- Delegating: "Author commits with `git-commits`."
|
||||
|
||||
The same holds for a procedure another agent owns. What belongs in the body is what no invocable
|
||||
skill covers: the agent's role, its boundaries, the order it works in, and the format it returns.
|
||||
|
||||
**State a read-only boundary in prose, not only in frontmatter.** `disallowedTools` denies the
|
||||
tools it names and nothing else — never `Bash`, which an agent with no `tools` field inherits — so
|
||||
an agent fenced only in frontmatter can still write through a shell redirect.
|
||||
|
||||
## Invocation axis
|
||||
|
||||
Decide before writing the description whether the agent is model-delegated (the runtime picks it)
|
||||
or reached only by name (`@agent-<name>`).
|
||||
|
||||
Only Copilot's cloud/IDE format expresses that in frontmatter: `disable-model-invocation: true`
|
||||
requires explicit invocation, and `user-invocable: false` hides an agent from manual invocation.
|
||||
Both live in `.github/copilot/agents/<name>.md` and are inert in the CLI format. Claude Code has
|
||||
no equivalent field, and neither does the vendor-neutral plugin/APM file, so at those scopes a
|
||||
name-invoked agent still needs a description precise enough not to steal delegations — the
|
||||
boundary clause is doing that work.
|
||||
|
||||
## One gate, two measurements
|
||||
|
||||
| Gate | SUGGESTION | FAIL | Counts |
|
||||
|---|---|---|---|
|
||||
| description | 250 chars | 400 chars | the `description:` value only |
|
||||
| body (Copilot limit) | 30,000 chars | — | the body only; content past it is truncated silently |
|
||||
|
||||
The 30,000-character Copilot ceiling is a runtime truncation limit, not a quality target, and it
|
||||
applies to a plugin/APM file too — that file compiles into a real Copilot agent downstream. An
|
||||
agent body long enough to approach it has a delegation defect, not a length problem.
|
||||
@@ -0,0 +1,93 @@
|
||||
---
|
||||
source_keys:
|
||||
- context7-websites-code-claude
|
||||
- claude-code-subagents-docs
|
||||
- github-plugins-creating
|
||||
---
|
||||
|
||||
# Creating a new agent
|
||||
|
||||
Return to `SKILL.md` Step 4 once Step 3 below is done — validation, the version bump and commit
|
||||
verification are shared with the improve flow and are not repeated here.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Before touching the filesystem, confirm you have:
|
||||
|
||||
- [ ] Agent name (kebab-case, e.g. `code-reviewer`)
|
||||
- [ ] Root directory — a path inside a package for plugin/APM scope, a project root, or `~` for
|
||||
user scope
|
||||
- [ ] Agent purpose — one sentence describing the task this agent handles
|
||||
- [ ] Trigger condition — when should the runtime delegate to this agent?
|
||||
|
||||
If any are missing, stop and ask before proceeding.
|
||||
|
||||
`agent-audit` runs the validation in `SKILL.md` Step 4. It ships with the kyberforge plugin and
|
||||
is co-installed with this skill; if it is unavailable, stop and ask the user to install
|
||||
kyberforge before continuing.
|
||||
|
||||
Design for one job per agent. An agent covering two jobs is delegated to for the wrong one.
|
||||
|
||||
## Step 1 — Scaffold
|
||||
|
||||
```bash
|
||||
bash scripts/new-agent.sh <name> <root>
|
||||
```
|
||||
|
||||
Examples:
|
||||
|
||||
```bash
|
||||
bash scripts/new-agent.sh code-reviewer packages/my-package/ # plugin/APM scope if packages/my-package/apm.yml has a type: field
|
||||
bash scripts/new-agent.sh deploy-assistant .
|
||||
bash scripts/new-agent.sh security-reviewer ~
|
||||
```
|
||||
|
||||
The script resolves scope itself and prints which one it used and every path it wrote — read that
|
||||
output rather than predicting it. It walks up from `<root>` for the nearest ancestor `apm.yml`
|
||||
carrying a top-level `type:` field (`instructions`/`skill`/`hybrid`/`prompts`), which marks a
|
||||
package root and means plugin/APM scope. An `apm.yml` with no `type:` is a marketplace-only
|
||||
manifest: the walk skips it and keeps going. With no such manifest found, `<root>` being exactly
|
||||
`~` (checked directly, no walk-up) is user scope and anything else is project scope. A bare
|
||||
`plugin.json` no longer signals plugin scope — that path was replaced outright, not made
|
||||
dual-mode, and falls through to project scope.
|
||||
|
||||
The script is file-by-file no-op: it skips any file that already exists, so re-running it to
|
||||
complete a partial pair is safe.
|
||||
|
||||
If the script or a template under `assets/templates/` has to change to support this agent — a new
|
||||
scope, a new scaffolded field, different output — read `references/scripts.md` first. Its
|
||||
conventions are asserted by `tests/new-agent.bats`, and an edit that ignores them fails the suite.
|
||||
|
||||
## Step 2 — Fill in the file(s)
|
||||
|
||||
Take the scope the script reported and read the matching reference — `SKILL.md` Step 2 has the
|
||||
table. That file carries the field rules and the pre-audit checklist for this scope; the other one
|
||||
describes fields this run cannot use.
|
||||
|
||||
Every scaffolded file, at every scope:
|
||||
|
||||
1. Replace each `FILL IN:` placeholder.
|
||||
2. Delete every `<!-- ... -->` template comment from the frontmatter. `apm compile` copies plugin
|
||||
frontmatter verbatim and HTML comments are not valid YAML, so a leftover comment breaks the
|
||||
file downstream on both harnesses.
|
||||
3. Write the `description` against `references/contract.md` and the system prompt body against its
|
||||
Body section.
|
||||
|
||||
## Step 3 — Populate or delete `sources.md`
|
||||
|
||||
Plugin/APM scope only — skip at project and user scope, which have no package root to hold the
|
||||
file.
|
||||
|
||||
The scaffold writes a commented `sources.md` skeleton at the package root, alongside `apm.yml` and
|
||||
not inside `.apm/agents/`, so that tooling scanning that directory for agent definitions does not
|
||||
treat it as an agent missing its frontmatter (ADR-0010).
|
||||
|
||||
If a research `sources.md` is present in the conversation context, filter it to entries with
|
||||
`` `extracted` `` status, work out which agent file each one contributed to, and fill in the
|
||||
skeleton following the commented format already in the file. Paths in `Contributing files:` are
|
||||
relative to the package root. Each slug must match an H2 heading and must also appear in the
|
||||
`source_keys` list of every file named under its `Contributing files:`.
|
||||
|
||||
If no research sources are in context, delete `sources.md`.
|
||||
|
||||
Then return to `SKILL.md` Step 4.
|
||||
@@ -22,11 +22,14 @@ Agent definitions deploy at three scopes and behave differently at each. The sco
|
||||
|
||||
When the same agent `name` appears at multiple scopes, **user scope wins over project scope wins over plugin scope** in Claude Code. In Copilot CLI, repo-level agents override enterprise and org-level; home-directory (user) agents override repo-level on name collision.
|
||||
|
||||
## Plugin scope restrictions
|
||||
## Which fields exist where
|
||||
|
||||
Plugin/APM agents (`.apm/agents/<name>.agent.md`) carry only `name`, `description`, optionally `model`, and optionally `source_keys` (provenance metadata, not a runtime field — silently ignored by both harnesses) in frontmatter — full stop (see ADR-0016). `apm compile` copies this frontmatter verbatim to both the Claude Code and Copilot CLI compile targets with no per-target integrator: Claude's `tools:` (space-separated string) and Copilot's `tools:` (alias list) are incompatible vocabularies, and Claude-only fields have no Copilot equivalent, so any harness-specific value is guaranteed wrong on at least one target.
|
||||
|
||||
This makes the old "silently ignored at plugin scope" framing moot. It's not that `hooks`, `mcpServers`, `permissionMode`, `tools`, `isolation`, `maxTurns`, `effort`, `memory`, `disallowedTools`, `skills`, `color`, `initialPrompt`, or `background` are merely ignored at this scope — they are never written to the file at all. Copy the agent to `.claude/agents/` (project scope) or `~/.claude/agents/` (user scope) to use any of them.
|
||||
Field rules are per scope and live with the scope: `references/plugin-scope.md` for the single
|
||||
vendor-neutral file, `references/project-user-scope.md` for the Claude Code / Copilot pair. Read
|
||||
one, not both. The short version is that plugin/APM frontmatter is an allowlist read from
|
||||
`agent-audit`'s `references/field-inventory.md`, narrow because `apm compile` copies frontmatter
|
||||
verbatim to every target (ADR-0016), while project and user scope carry the full per-provider
|
||||
field sets.
|
||||
|
||||
## Scoped identifiers (Claude Code plugin agents only)
|
||||
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
---
|
||||
source_keys:
|
||||
- claude-code-subagents-docs
|
||||
---
|
||||
|
||||
# Improving an existing agent
|
||||
|
||||
Return to `SKILL.md` Step 4 once Step 4 below is done — validation, the version bump and commit
|
||||
verification are shared with the create flow and are not repeated here.
|
||||
|
||||
## Step 1 — Verify inputs
|
||||
|
||||
Confirm the agent file (or, at project and user scope, the pair) exists and that at least one
|
||||
improvement signal is present in the conversation or in a referenced file.
|
||||
|
||||
If no signals are present, stop: "This skill applies existing signals to an agent. For a blind
|
||||
review, run `agent-audit` instead."
|
||||
|
||||
`agent-audit` runs the validation in `SKILL.md` Step 4 and is co-installed with this skill; if
|
||||
it is unavailable, stop and ask the user to install the kyberforge plugin before continuing.
|
||||
|
||||
**Partial pair — project and user scope only.** If one provider file exists and the other does
|
||||
not, scaffold the missing one with `bash scripts/new-agent.sh <name> <root>` (file-by-file no-op)
|
||||
and continue. Plugin/APM scope is a single file and has no partial state.
|
||||
|
||||
## Step 2 — Gather and group signals
|
||||
|
||||
Read the current file(s), then collect every signal from the conversation and from any path the
|
||||
user referenced.
|
||||
|
||||
Group signals by **root cause**, not by symptom. Patching per symptom is the default failure mode:
|
||||
three complaints often trace to one missing instruction. Ask: "What single gap in this agent
|
||||
causes this cluster?" One root cause, one fix.
|
||||
|
||||
```text
|
||||
Example:
|
||||
- User feedback: the agent keeps trying to push to the remote
|
||||
- Session context: no scope boundary in the system prompt
|
||||
→ Root cause: the system prompt has no git scope constraint → fix: add an explicit boundary
|
||||
```
|
||||
|
||||
## Step 3 — Announce planned changes
|
||||
|
||||
Before editing, state which root causes were identified, what evidence supports each, and which
|
||||
files will change. Then proceed — edits are reversible via git, so no approval checkpoint is
|
||||
needed.
|
||||
|
||||
## Step 4 — Apply changes
|
||||
|
||||
Edit whichever file the signals point to.
|
||||
|
||||
**Generalize, do not patch.** Fix the underlying gap, not the one example that failed. A fix
|
||||
scoped to the cases you have seen overfits and performs worse on new input.
|
||||
|
||||
**Delegate rather than grow.** An agent body has no word ceiling, but a body that restates a
|
||||
procedure a skill it can invoke already owns is an `agent-audit` FAIL. When a signal reports a
|
||||
missing procedure, check first whether an installed skill owns it and name that skill instead of
|
||||
transcribing it. See `references/contract.md`.
|
||||
|
||||
The delegation check is not a length brake — it fires only on procedure an invocable skill already
|
||||
owns, and says nothing about original prose. That brake is judgment, and it is the only one left:
|
||||
for every sentence you add, ask "would the agent get this wrong without it?" and delete it if the
|
||||
answer is no.
|
||||
|
||||
**Explain the why.** Reasoning-based instructions outperform rigid directives. A rule written in
|
||||
all caps (ALWAYS/NEVER) is usually better reframed as why the behaviour matters, so the agent can
|
||||
apply judgment at the edges.
|
||||
|
||||
**Retrofit before extending.** Any agent predating ADR-0020 has to meet the description contract
|
||||
before any other edit lands — the gates are hot and carry no baseline file, so a one-line fix to a
|
||||
non-compliant agent cannot be committed until its description meets `references/contract.md`.
|
||||
Treat that retrofit as part of the same change, not a follow-up.
|
||||
|
||||
**Re-check the scope rules.** Read the reference for the resolved scope (`SKILL.md` Step 2) and
|
||||
confirm the edit introduced no field that scope forbids, and dropped no `disallowedTools` fence
|
||||
that was already there.
|
||||
|
||||
If the edit adds or removes research-sourced content, update `source_keys` in the edited file and
|
||||
the matching `sources.md` entry — the create flow's Step 3 has the rules.
|
||||
|
||||
**Check for regressions before handing back.** `SKILL.md` Step 4 tells you to resolve every FAIL,
|
||||
which says nothing about a check that passed *before* these edits and no longer does. Compare the
|
||||
closing `agent-audit` against the agent's pre-edit state — a PASS that has become a SUGGESTION, or
|
||||
a SUGGESTION that has become a FAIL, is damage this flow caused and is in scope for it. Only the
|
||||
improve flow can make that comparison; the create flow has no prior state to compare against.
|
||||
|
||||
Then return to `SKILL.md` Step 4.
|
||||
@@ -0,0 +1,74 @@
|
||||
---
|
||||
source_keys:
|
||||
- context7-websites-code-claude
|
||||
- claude-code-plugins-docs
|
||||
---
|
||||
|
||||
# Plugin/APM scope — the single vendor-neutral file
|
||||
|
||||
One file, no counterpart: `<package-root>/.apm/agents/<name>.agent.md`. `apm compile` emits it to
|
||||
both the Claude Code and the Copilot CLI target. The `.agent.md` extension here is convention, not
|
||||
a Copilot marker — the file is vendor-neutral.
|
||||
|
||||
## Frontmatter
|
||||
|
||||
The permitted keys are the `apm-agent-allowlist` section of `agent-audit`'s
|
||||
`references/field-inventory.md`. Read them from there as data — that section is the single source
|
||||
of truth, `agent-audit`'s `validate.sh` parses it at load time, and it changes. Any restatement of
|
||||
the roster, here or in a template or in script output, goes stale one step further out than the
|
||||
list itself.
|
||||
|
||||
- `name` — kebab-case, must equal the filename stem, unique within the scope.
|
||||
- `description` — write it against `references/contract.md`.
|
||||
- Everything else — check the allowlist section before adding a key. A key outside it fails the
|
||||
audit.
|
||||
|
||||
**Why the list is narrow.** `apm compile` copies frontmatter verbatim to every target with no
|
||||
per-target integrator, so a harness-specific value is wrong on at least one of them (ADR-0016).
|
||||
The rule is about a field's *shape*, not a fixed roster:
|
||||
|
||||
- `tools` is an **allowlist** whose vocabulary differs per harness — Claude Code names its own
|
||||
tools, Copilot CLI uses aliases (`execute`/`read`/`edit`/`search`/`agent`/`web`) — so one value
|
||||
is wrong on one target. It stays out. Omitting it means inherit-all-tools on both, which is
|
||||
never wrong.
|
||||
- `disallowedTools` is a **denylist**, and denying by name cannot fail that way: a name the other
|
||||
harness does not recognise denies nothing, so the worst case is a missing fence, never a wrongly
|
||||
granted capability. That asymmetry is the whole exception (ADR-0016's 2026-08-14 amendment).
|
||||
Claude Code honours it for plugin subagents; the three fields plugin agents do silently ignore
|
||||
are `hooks`, `mcpServers` and `permissionMode`, and this is not one of them. Copilot's handling
|
||||
of the key is unconfirmed, which ADR-0016 accepts as a stated risk.
|
||||
|
||||
Its syntax is the same at every scope, and this is the one scope that cannot reach it anywhere
|
||||
else: MCP tools are denied as `mcp__<server>`, `mcp__<server>__*` or `mcp__*`; both a YAML list
|
||||
and a delimited string are accepted, and this repo writes the comma-separated string form
|
||||
(`disallowedTools: Edit, Write, NotebookEdit`) — match it.
|
||||
- The Claude-only knobs (`isolation`, `maxTurns`, `effort`, `memory`, `permissionMode`, `skills`,
|
||||
`color`, `initialPrompt`, `background`, `hooks`, `mcpServers`) have no Copilot equivalent and
|
||||
are never written to this file at all. "Silently ignored at plugin scope" is the wrong framing:
|
||||
they are absent, not tolerated. To use any of them, copy the agent to `.claude/agents/`
|
||||
(project scope) or `~/.claude/agents/` (user scope).
|
||||
|
||||
Write `disallowedTools` on every read-only plugin-scope agent — and say the agent is read-only in
|
||||
the body as well, because the fence does not cover the inherited `Bash` tool.
|
||||
|
||||
`source_keys` is provenance metadata, not a runtime field: both harnesses ignore it. Add it only
|
||||
when research sources informed the agent, with slugs matching H2 headings in the package root's
|
||||
`sources.md`.
|
||||
|
||||
## Body
|
||||
|
||||
Follow the Body section of `references/contract.md`: role instruction, one job, and delegation
|
||||
to installed skills instead of transcribed procedure.
|
||||
|
||||
## Before invoking `agent-audit`
|
||||
|
||||
- [ ] `name` kebab-case, matching the filename stem, unique in scope
|
||||
- [ ] `description` written to `references/contract.md`
|
||||
- [ ] Every frontmatter key present in the `apm-agent-allowlist` section — in particular no `tools`
|
||||
- [ ] No `FILL IN:` placeholder and no `<!-- ... -->` template comment anywhere in the file
|
||||
- [ ] System prompt body non-empty, and a read-only agent says so in prose as well as in
|
||||
`disallowedTools`
|
||||
- [ ] Body covers all four required elements: inputs expected, process steps, output format,
|
||||
**error handling** — what the agent does on malformed, missing or contradictory input
|
||||
|
||||
Then return to the flow reference you came from.
|
||||
@@ -0,0 +1,111 @@
|
||||
---
|
||||
source_keys:
|
||||
- claude-code-subagents-docs
|
||||
- context7-github-en-copilot
|
||||
- github-custom-agents-configuration
|
||||
- github-cli-plugin-reference
|
||||
---
|
||||
|
||||
# Project and user scope — the Claude Code / Copilot pair
|
||||
|
||||
Two files per agent, written in one pass and kept in step: a Claude Code `.md` and a Copilot CLI
|
||||
`.agent.md`. The system prompt body is the same in both — the agent's task does not change with
|
||||
the provider. The frontmatter is not.
|
||||
|
||||
| Scope | Claude Code | Copilot CLI |
|
||||
|---|---|---|
|
||||
| Project | `.claude/agents/<name>.md` | `.github/agents/<name>.agent.md` |
|
||||
| User | `~/.claude/agents/<name>.md` | `~/.copilot/agents/<name>.agent.md` |
|
||||
|
||||
## Claude Code file
|
||||
|
||||
**`name`** — lowercase letters and hyphens only, unique within the scope. Claude Code discards a
|
||||
duplicate silently.
|
||||
|
||||
**`description`** — write it against `references/contract.md`. It is the primary signal for
|
||||
autonomous delegation.
|
||||
|
||||
**`tools`** — an allowlist. Write it, and restrict it to the tools the agent actually needs;
|
||||
omitting it inherits every tool from the parent, which is the right value only when the agent
|
||||
genuinely needs all of them. Least privilege is the default, not the exception. Use
|
||||
`Agent(type1,type2)`
|
||||
to restrict which subagent types this agent may spawn, and omit `Agent` entirely to stop it
|
||||
spawning any. Five tools reach no subagent whatever this field says — `AskUserQuestion`,
|
||||
`EnterPlanMode`, `ExitPlanMode`, `ScheduleWakeup` and `WaitForMcpServers` — so listing one buys
|
||||
nothing. The single exception is `ExitPlanMode`, available when the parent session runs
|
||||
`permissionMode: plan`.
|
||||
|
||||
**`disallowedTools`** — a denylist, applied before `tools` and taking precedence over it. Supports
|
||||
`mcp__<server>`, `mcp__<server>__*` and `mcp__*` globs. Both a YAML list and a delimited string
|
||||
are accepted; this repo writes the comma-separated string form (`disallowedTools: Edit, Write,
|
||||
NotebookEdit`) — match it.
|
||||
|
||||
**`model`** — set it when the agent needs a different capability tier (`haiku` for fast lookups,
|
||||
`opus` for deep reasoning). Resolution order is `CLAUDE_CODE_SUBAGENT_MODEL` → the per-invocation
|
||||
parameter → this field → the main session model, so the frontmatter value is a low-priority
|
||||
default rather than a guarantee.
|
||||
|
||||
Optional fields worth considering, none of which exist at plugin/APM scope:
|
||||
|
||||
- `maxTurns` — cap agentic turns on a bounded task, to stop a runaway
|
||||
- `effort` — `low` for a single lookup, `high` or above for multi-file analysis; omit to inherit
|
||||
- `memory` — `user`, `project` or `local`; only when cross-session state is genuinely needed
|
||||
- `isolation: worktree` — only when the agent modifies files and needs an isolated copy
|
||||
- `skills` — skill names preloaded at agent startup; unrelated to the `source_keys` metadata field
|
||||
- `color` — the UI tile colour (`red`, `blue`, `green`, `yellow`, `purple`, `orange`, `pink`,
|
||||
`cyan`)
|
||||
- `background` — `true` forces background execution
|
||||
- `initialPrompt` — auto-submitted as the first turn when the agent activates as the main session
|
||||
thread; set it only for a main-thread agent, never for a subagent
|
||||
|
||||
`hooks`, `mcpServers` and `permissionMode` are honoured at these two scopes and nowhere else — a
|
||||
plugin agent carrying them is ignored silently.
|
||||
|
||||
A subdirectory under `agents/` does not affect the agent's name at these scopes; it does at plugin
|
||||
scope, which is one reason `references/deployment-modes.md` recommends keeping agents flat.
|
||||
|
||||
## Copilot file
|
||||
|
||||
Two Copilot formats exist, with different paths and different field sets. Pick one:
|
||||
|
||||
**CLI format** — what the scaffold writes.
|
||||
|
||||
- Path: `.github/agents/<name>.agent.md` (project) or `~/.copilot/agents/<name>.agent.md` (user)
|
||||
- The `.agent.md` extension is mandatory: Copilot CLI does not pick up a plain `.md` file in
|
||||
`agents/`, and fails silently rather than reporting it
|
||||
- Fields: `name` (required, must equal the filename stem), `description` (required), `tools`
|
||||
(optional)
|
||||
- `tools` uses Copilot aliases, not Claude tool names: `execute` (shell), `read`, `edit`,
|
||||
`search`, `agent`, `web`; MCP tools as `server-name/tool-name` or `server-name/*`
|
||||
|
||||
**Cloud/IDE format** — for Copilot Chat in VS Code or on GitHub.com.
|
||||
|
||||
- Path: `.github/copilot/agents/<name>.md` — a plain `.md`, in a different directory
|
||||
- Adds `target` (`vscode`, `github-copilot`, or omit for both), `user-invocable`,
|
||||
`disable-model-invocation` and `mcp-servers` (processed by the cloud runtime, ignored in VS
|
||||
Code). These four are inert in the CLI format — do not write them there
|
||||
- This is the only format that can express the invocation axis in frontmatter; see the Invocation
|
||||
axis section of `references/contract.md`
|
||||
|
||||
Both formats truncate a body past **30,000 characters** silently.
|
||||
|
||||
Copilot has no `permissionMode`, `maxTurns`, `isolation`, `memory`, `effort`, `hooks` or
|
||||
`mcpServers`. Never let those cross over from the Claude Code file.
|
||||
|
||||
## Before invoking `agent-audit`
|
||||
|
||||
Both files:
|
||||
|
||||
- [ ] `name` present and kebab-case; `description` written to `references/contract.md`
|
||||
- [ ] System prompt body present, non-empty and equivalent across the pair
|
||||
- [ ] Body covers all four required elements: inputs expected, process steps, output format,
|
||||
**error handling** — what the agent does on malformed, missing or contradictory input
|
||||
- [ ] No `FILL IN:` placeholder and no `<!-- ... -->` template comment left
|
||||
|
||||
Copilot file only:
|
||||
|
||||
- [ ] Extension is `.agent.md` (CLI format), and `name` matches the filename stem
|
||||
- [ ] No Claude Code-only field present
|
||||
- [ ] Body under 30,000 characters
|
||||
|
||||
Then return to the flow reference you came from.
|
||||
@@ -15,6 +15,7 @@ All scripts in this skill must follow these rules:
|
||||
- **Idempotent** — "create if not exists" per file. The scaffold script skips any file that already exists; agents may safely re-run it.
|
||||
- **Meaningful exit codes** — `0` success, `1` invalid arguments or precondition failure. Document in `--help`.
|
||||
- **Self-contained** — no external package installs at runtime. The script uses only bash builtins and POSIX tools (`sed`, `mkdir`, `cat`).
|
||||
- **No restated field rosters** — no script output, in `--help` or in next-steps guidance, enumerates permitted, forbidden, or required frontmatter fields. Point at the `apm-agent-allowlist` section of `agent-audit`'s `references/field-inventory.md`, which `agent-audit`'s `validate.sh` reads from there as data. A roster copied into script output goes stale one step further out than the list itself: the next-steps hint `(name, description, model, body only)` kept printing after ADR-0016's 2026-08-14 amendment added `disallowedTools` to the permitted set. `tests/new-agent.bats` enforces this for the plugin/APM branch — naming some allowlisted fields but not all is a failure.
|
||||
|
||||
## Template variables
|
||||
|
||||
|
||||
@@ -19,7 +19,7 @@ source_keys:
|
||||
- **URL:** context7:/websites/code_claude
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/claude-code-plugins/sources.md
|
||||
- **Description:** Official Claude Code documentation site indexed by Context7 — plugin manifest schema, subagent definition types, marketplace JSON format, agent markdown file format
|
||||
- **Contributing files:** SKILL.md, references/deployment-modes.md
|
||||
- **Contributing files:** SKILL.md, references/create.md, references/deployment-modes.md, references/plugin-scope.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## claude-code-plugins-docs
|
||||
@@ -27,7 +27,7 @@ source_keys:
|
||||
- **URL:** https://code.claude.com/docs/en/plugins
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/claude-code-plugins/sources.md
|
||||
- **Description:** Official Claude Code plugin authoring guide — plugin structure, manifest fields, loading methods, skill namespacing, agent activation, marketplace submission
|
||||
- **Contributing files:** SKILL.md, references/deployment-modes.md
|
||||
- **Contributing files:** SKILL.md, references/deployment-modes.md, references/plugin-scope.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## claude-code-subagents-docs
|
||||
@@ -35,7 +35,7 @@ source_keys:
|
||||
- **URL:** https://code.claude.com/docs/en/sub-agents
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/claude-code-plugins/sources.md
|
||||
- **Description:** Official Claude Code subagent reference — definition format, all frontmatter fields, scope priority, built-in agents, CLI flags, environment variables, known limitations
|
||||
- **Contributing files:** SKILL.md, references/deployment-modes.md
|
||||
- **Contributing files:** SKILL.md, references/create.md, references/improve.md, references/contract.md, references/deployment-modes.md, references/project-user-scope.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## context7-github-en-copilot
|
||||
@@ -43,7 +43,7 @@ source_keys:
|
||||
- **URL:** context7:/websites/github_en_copilot
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/github-copilot-plugins/sources.md
|
||||
- **Description:** Official GitHub Copilot documentation indexed by Context7; covers CLI plugins, custom agents, SDK, and marketplace
|
||||
- **Contributing files:** SKILL.md, references/deployment-modes.md
|
||||
- **Contributing files:** references/deployment-modes.md, references/project-user-scope.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## github-custom-agents-configuration
|
||||
@@ -51,7 +51,7 @@ source_keys:
|
||||
- **URL:** https://docs.github.com/en/copilot/reference/custom-agents-configuration
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/github-copilot-plugins/sources.md
|
||||
- **Description:** Reference for cloud and IDE custom agent definition format — frontmatter fields, tool aliases, MCP server config, secrets interpolation, scoping hierarchy
|
||||
- **Contributing files:** SKILL.md, references/deployment-modes.md
|
||||
- **Contributing files:** references/contract.md, references/deployment-modes.md, references/project-user-scope.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## github-cli-plugin-reference
|
||||
@@ -59,7 +59,7 @@ source_keys:
|
||||
- **URL:** https://docs.github.com/en/copilot/reference/copilot-cli-reference/cli-plugin-reference
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/github-copilot-plugins/sources.md
|
||||
- **Description:** Full CLI plugin reference — plugin.json schema, marketplace.json schema, all CLI commands and flags, install specification formats, loading precedence, env vars, LSP config
|
||||
- **Contributing files:** SKILL.md, references/deployment-modes.md
|
||||
- **Contributing files:** references/deployment-modes.md, references/project-user-scope.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## github-plugins-creating
|
||||
@@ -67,7 +67,7 @@ source_keys:
|
||||
- **URL:** https://docs.github.com/en/copilot/how-tos/copilot-cli/customize-copilot/plugins-creating
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/github-copilot-plugins/sources.md
|
||||
- **Description:** How-to for creating Copilot CLI plugins — plugin structure, agent and skill authoring, hooks format, MCP config, development lifecycle
|
||||
- **Contributing files:** SKILL.md
|
||||
- **Contributing files:** references/create.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## github-plugins-finding-installing
|
||||
|
||||
@@ -10,4 +10,4 @@ Usage: new-agent.sh <agent-name> <root>
|
||||
|
||||
Resolves scope by walking up from `<root>`: a `type:`-bearing `apm.yml` found at or above `<root>` → plugin/APM scope (single file at `<package-root>/.apm/agents/<name>.agent.md`; an `apm.yml` without `type:` is a marketplace-only manifest and is skipped); `<root>` exactly `~` → user scope (`~/.claude/agents/` + `~/.copilot/agents/`); otherwise project scope (`<root>/.claude/agents/` + `<root>/.github/agents/`). Each file is a no-op if it already exists. See `--help` for full usage.
|
||||
|
||||
Tests: `tests/new-agent.bats` (requires `bats-support` and `bats-assert`).
|
||||
Tests: `tests/new-agent.bats` (requires `bats-support` and `bats-assert`) — source-only. `scripts/sync-plugin-content.sh` strips `<category>/<name>/tests` from the generated mirror (ADR-0017), so this file exists in a repo checkout of `.apm/skills/agent-author/` and not in an installed plugin.
|
||||
|
||||
@@ -21,10 +21,12 @@ Arguments:
|
||||
without type: is a marketplace-only manifest
|
||||
and is skipped, the walk continues upward
|
||||
→ creates <package-root>/.apm/agents/<name>.agent.md
|
||||
(single vendor-neutral file — no tools,
|
||||
isolation, maxTurns, effort, memory, or
|
||||
permissionMode; apm compile has no per-target
|
||||
field integrator, see ADR-0016)
|
||||
(single vendor-neutral file; apm compile copies
|
||||
its frontmatter verbatim to every target with no
|
||||
per-target field integrator, so the permitted
|
||||
field set is narrow — see the apm-agent-allowlist
|
||||
section of agent-audit's
|
||||
references/field-inventory.md and ADR-0016)
|
||||
→ creates <package-root>/sources.md (if absent)
|
||||
project scope : no type:-bearing apm.yml found; root is a
|
||||
project directory
|
||||
@@ -155,9 +157,11 @@ find_package_root() {
|
||||
done
|
||||
}
|
||||
|
||||
# `read` consumes a single line, so kind and path are emitted on one
|
||||
# space-separated line rather than two `echo`s — kind first (never contains
|
||||
# spaces), path last (absorbs any spaces in the path safely).
|
||||
# kind and path are emitted on one space-separated line rather than two
|
||||
# `echo`s — kind first (never contains spaces), path last (absorbs any spaces
|
||||
# in the path safely). `mapfile`/`readarray` would need bash 4.0+, which
|
||||
# macOS's stock /bin/bash 3.2 is not; a here-string `read` splits the single
|
||||
# line without it. Same form as skill-author's new-skill.sh, deliberately.
|
||||
WALK_RESULT="$(find_package_root "$ROOT")"
|
||||
read -r WALK_KIND WALK_ROOT <<< "$WALK_RESULT"
|
||||
|
||||
@@ -258,6 +262,19 @@ SOURCES
|
||||
fi
|
||||
fi
|
||||
|
||||
# Next-steps guidance names no frontmatter fields, by rule (see references/scripts.md).
|
||||
# A roster restated in terminal output goes stale one step further out than the list
|
||||
# itself: the old "(name, description, model, body only)" hint outlived ADR-0016's
|
||||
# 2026-08-14 amendment, which added disallowedTools to the permitted set. Point at the
|
||||
# scaffolded file's own comments for what to fill, and at agent-audit's validate.sh —
|
||||
# which reads the allowlist from field-inventory.md as data — for what is permitted.
|
||||
AUDIT_SCRIPTS="$(cd "$SKILL_ROOT/../agent-audit/scripts" 2>/dev/null && pwd || true)"
|
||||
if [[ -n "$AUDIT_SCRIPTS" && -f "$AUDIT_SCRIPTS/validate.sh" ]]; then
|
||||
VALIDATE_HINT="$AUDIT_SCRIPTS/validate.sh"
|
||||
else
|
||||
VALIDATE_HINT="agent-audit's scripts/validate.sh"
|
||||
fi
|
||||
|
||||
if [[ "$created_any" == false ]]; then
|
||||
echo "All files already exist — nothing to do." >&2
|
||||
else
|
||||
@@ -266,12 +283,21 @@ else
|
||||
echo "" >&2
|
||||
echo "Next steps:" >&2
|
||||
if [[ "$SCOPE" == "plugin" ]]; then
|
||||
echo " 1. Fill in $APM_FILE — replace all FILL IN: placeholders (name, description, model, body only)" >&2
|
||||
echo " 1. Fill in $APM_FILE — replace every FILL IN: placeholder. Optional fields are" >&2
|
||||
echo " scaffolded there as commented blocks; uncomment the ones that apply." >&2
|
||||
echo " Description: 250 chars target / 400 ceiling (ADR-0020). The body has no" >&2
|
||||
echo " word gate — delegate to a skill instead of restating what it does." >&2
|
||||
echo " 2. Populate $SOURCES_DIR/sources.md with research sources, or delete it" >&2
|
||||
echo " 3. Validate: check required fields (name, description, system prompt) in the file" >&2
|
||||
echo " 3. Validate: $VALIDATE_HINT $APM_FILE" >&2
|
||||
echo " It checks the frontmatter against the apm-agent-allowlist section of" >&2
|
||||
echo " agent-audit's references/field-inventory.md, the authoritative field list." >&2
|
||||
else
|
||||
echo " 1. Fill in $CC_FILE — replace all FILL IN: placeholders" >&2
|
||||
echo " 2. Fill in $CP_FILE — replace all FILL IN: placeholders" >&2
|
||||
echo " 3. Validate: check required fields (name, description, system prompt) in both files" >&2
|
||||
echo " 1. Fill in $CC_FILE — replace every FILL IN: placeholder. Optional fields are" >&2
|
||||
echo " scaffolded there as commented blocks; uncomment the ones that apply." >&2
|
||||
echo " Description: 250 chars target / 400 ceiling (ADR-0020). The body has no" >&2
|
||||
echo " word gate — delegate to a skill instead of restating what it does." >&2
|
||||
echo " 2. Fill in $CP_FILE — same, and heed its closing comment: the Claude Code-only" >&2
|
||||
echo " fields it names must not cross over from the file above." >&2
|
||||
echo " 3. Validate: run $VALIDATE_HINT on each file" >&2
|
||||
fi
|
||||
fi
|
||||
|
||||
@@ -77,19 +77,67 @@ teardown() {
|
||||
assert_failure
|
||||
}
|
||||
|
||||
@test "plugin/APM scope: frontmatter carries only name, description, model, source_keys fields" {
|
||||
# The permitted set is read from the same data agent-audit's validate.sh reads --
|
||||
# the apm-agent-allowlist section of agent-audit's field-inventory.md -- rather than
|
||||
# restated here. A hardcoded copy drifts: this assertion listed four fields and went
|
||||
# on passing after ADR-0016's amendment added disallowedTools, and would have
|
||||
# rejected a scaffolded agent that legitimately carried it.
|
||||
@test "plugin/APM scope: frontmatter carries only allowlisted fields" {
|
||||
inventory="$BATS_TEST_DIRNAME/../../agent-audit/references/field-inventory.md"
|
||||
[ -f "$inventory" ] || fail "field-inventory.md not found at $inventory"
|
||||
allowlist="$(awk '
|
||||
/^## apm-agent-allowlist$/ { insection = 1; next }
|
||||
insection && /^##/ { exit }
|
||||
insection && NF && $0 !~ /^#/ && $0 !~ /^---/ { print; exit }
|
||||
' "$inventory")"
|
||||
[ -n "$allowlist" ] || fail "apm-agent-allowlist section is empty in $inventory"
|
||||
|
||||
printf 'name: my-package\ntype: skill\n' > "$ROOT/apm.yml"
|
||||
bash "$SCRIPT" my-agent "$ROOT"
|
||||
file="$ROOT/.apm/agents/my-agent.agent.md"
|
||||
fm="$(sed -n '/^---$/,/^---$/p' "$file")"
|
||||
keys="$(grep -oE '^[a-zA-Z][a-zA-Z0-9_-]*:' <<< "$fm" | sed 's/:$//' | sort -u)"
|
||||
for key in $keys; do
|
||||
if [[ "$key" != "name" && "$key" != "description" && "$key" != "model" && "$key" != "source_keys" ]]; then
|
||||
fail "unexpected frontmatter key: $key"
|
||||
if ! grep -qw "$key" <<< "$allowlist"; then
|
||||
fail "frontmatter key '$key' is not in field-inventory.md's apm-agent-allowlist ($allowlist)"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
# Same drift class as the assertion above, one step further out: the next-steps text
|
||||
# used to enumerate "(name, description, model, body only)" and went on printing that
|
||||
# after ADR-0016's amendment added disallowedTools. Rather than assert on wording, this
|
||||
# asserts the roster is all-or-nothing: if the guidance names any allowlisted field it
|
||||
# must name every one of them, so a partial restatement -- the only shape that can go
|
||||
# stale silently -- fails. Naming none, the current design, passes.
|
||||
@test "plugin/APM scope: next-steps guidance does not partially restate the allowlist" {
|
||||
inventory="$BATS_TEST_DIRNAME/../../agent-audit/references/field-inventory.md"
|
||||
[ -f "$inventory" ] || fail "field-inventory.md not found at $inventory"
|
||||
allowlist="$(awk '
|
||||
/^## apm-agent-allowlist$/ { insection = 1; next }
|
||||
insection && /^##/ { exit }
|
||||
insection && NF && $0 !~ /^#/ && $0 !~ /^---/ { print; exit }
|
||||
' "$inventory")"
|
||||
[ -n "$allowlist" ] || fail "apm-agent-allowlist section is empty in $inventory"
|
||||
|
||||
printf 'name: my-package\ntype: skill\n' > "$ROOT/apm.yml"
|
||||
steps="$(bash "$SCRIPT" my-agent "$ROOT" 2>&1 | sed -n '/^Next steps:/,$p')"
|
||||
[ -n "$steps" ] || fail "scaffolder printed no next-steps block"
|
||||
|
||||
named=""
|
||||
missing=""
|
||||
for field in $allowlist; do
|
||||
if grep -qw -- "$field" <<< "$steps"; then
|
||||
named="$named $field"
|
||||
else
|
||||
missing="$missing $field"
|
||||
fi
|
||||
done
|
||||
if [ -n "$named" ] && [ -n "$missing" ]; then
|
||||
fail "next-steps names allowlisted field(s)$named but omits$missing -- a partial roster. Point at field-inventory.md instead of restating it."
|
||||
fi
|
||||
}
|
||||
|
||||
@test "plugin/APM scope: sources.md contributing-files template mentions the single-file path" {
|
||||
printf 'name: my-package\ntype: skill\n' > "$ROOT/apm.yml"
|
||||
bash "$SCRIPT" my-agent "$ROOT"
|
||||
|
||||
@@ -1,13 +1,17 @@
|
||||
# skill-audit
|
||||
|
||||
Audit a skill directory against the agentskills.io specification. Runs structural validation then a qualitative review across description quality, body discipline, patterns, formatting, file structure, scripts, and internal consistency, plus a provenance chain check.
|
||||
Audit a skill directory against the agentskills.io specification and the house context-budget contract (ADR-0020). Runs structural validation then a qualitative review across description quality, body discipline, patterns, formatting, file structure, scripts, and internal consistency, plus a provenance chain check.
|
||||
|
||||
## What it does
|
||||
|
||||
1. Runs `scripts/validate.sh` and `scripts/validate-provenance.sh` for structural and provenance checks, plus `scripts/vale-wrap.sh` — a Vale prefilter that deterministically flags known-bad description openers, vague wording, padding phrases, and "There is/are" sentence openers
|
||||
1. Runs `scripts/validate.sh` and `scripts/validate-provenance.sh` for structural and provenance checks, plus `scripts/vale-wrap.sh` — a Vale prefilter that deterministically flags non-imperative description openers, composition and architecture notes, vague wording, padding phrases, and "There is/are" sentence openers
|
||||
2. Reads all files in the skill directory
|
||||
3. Applies qualitative checks across seven dimensions
|
||||
4. Outputs a compact findings report — findings only, grouped by dimension, each with Why and Fix — and a result block with handoff to /skill-improve
|
||||
3. Applies qualitative checks across five dimension groups, loading one rubric from `references/` per group
|
||||
4. Outputs a compact findings report — findings only, grouped by dimension, each with Why and Fix — and a result block with handoff to `skill-author`
|
||||
|
||||
`validate.sh` enforces two independent length families that must not be conflated: the agentskills.io spec conformance ceilings (500 lines, 2,770 words, both counting the whole file) and the ADR-0020 context budget (250/400 description characters, 600/900 body-only words).
|
||||
|
||||
Alongside those it runs four shape checks that are not length measurements at all. Two are FAILs: every routing target named in the description — in the compressed `Not <thing> -> <name>` arrow **and** in the prose form — must resolve to a real skill or agent, and every `references/<file>.md` the body names must exist on disk. Three are SUGGESTIONs: a missing boundary clause, a Gotchas section over five entries, and a Gotchas section over 25% of the body. The resolution universe for boundary targets is derived by walking up from the audited `SKILL.md` — the authoring root above it, its own apm package, and that package's declared `apm.yml` dependencies — so a fresh clone and a machine that has run `apm install` return the same verdict. When no universe can be determined the check prints `INFO ... DID NOT RUN` and does not silently pass.
|
||||
|
||||
## Usage
|
||||
|
||||
@@ -22,17 +26,27 @@ Provide the path to the skill directory to audit when invoking.
|
||||
| File | Purpose |
|
||||
|------|---------|
|
||||
| `SKILL.md` | Skill instructions for agents |
|
||||
| `scripts/validate.sh` | Structural validator — checks name format, name matches directory, description length, line count, placeholder detection, script executable bit, and interactive-prompt detection |
|
||||
| `scripts/validate.sh` | Structural validator — checks name format, name matches directory, description presence and length, body-only word count, line and whole-file word ceilings, boundary-clause presence, boundary-target resolution, `references/` pointer existence, Gotchas entry count and body share, placeholder detection, script executable bit, and interactive-prompt detection |
|
||||
| `scripts/validate-provenance.sh` | Provenance validator — checks sources.md completeness, source_keys/slug consistency, Contributing files existence, bidirectional linkage, Research doc: fields, and upstream research doc alignment |
|
||||
| `scripts/vale-wrap.sh` | Vale prefilter wrapper — runs the bundled `Kyberforge` Vale styles against SKILL.md and reports alerts as deterministic FAILs ahead of Step 3's qualitative review |
|
||||
| `assets/vale/.vale.ini` | Vale configuration — points Vale at the bundled `Kyberforge` style path, self-located relative to `vale-wrap.sh` |
|
||||
| `assets/vale/styles/Kyberforge/DescriptionOpener.yml` | Vale rule — flags literal "This skill..."/"This agent..." description openers |
|
||||
| `assets/vale/styles/Kyberforge/CompositionNote.yml` | Vale rule — flags composition and architecture notes in a description (e.g. "cross-cutting", "entry point", "rather than duplicating") |
|
||||
| `assets/vale/styles/Kyberforge/DescriptionOpener.yml` | Vale rule — flags non-imperative "This..." description openers |
|
||||
| `assets/vale/styles/Kyberforge/PaddingPhrase.yml` | Vale rule — flags generic "see references/" padding phrasing in conditional references |
|
||||
| `assets/vale/styles/Kyberforge/SentenceOpenerThereIs.yml` | Vale rule — flags body sentences starting with "There is"/"There are" |
|
||||
| `assets/vale/styles/Kyberforge/VagueWording.yml` | Vale rule — flags known filler wording (e.g. "helps with", "utilize") |
|
||||
| `references/description-quality.md` | Spec-grounded rubric for description auditing — loaded when a finding is borderline |
|
||||
| `references/body-discipline.md` | Spec-grounded rubric for body discipline auditing — loaded when padding vs necessity is unclear |
|
||||
| `references/description-quality.md` | Rubric for the description dimension — three-part shape, the 250/400-character budget, the hand-invoked (`disable-model-invocation`) contract, and the internal-mechanics FAIL |
|
||||
| `references/body-discipline.md` | Rubric for the body-discipline dimension — the core test, the 600/900 body-only budget against the 2,770-word whole-file backstop, the mandatory-dispatch rule, and the Gotchas constraints |
|
||||
| `references/patterns.md` | Rubric for the patterns dimension — which instruction construct fits which job, and how each is correctly formed |
|
||||
| `references/file-structure.md` | Rubric for the file-structure and internal-consistency dimensions — permitted directories, cross-plugin path rules and their two structural exemptions, README drift |
|
||||
| `references/formatting-and-scripts.md` | Rubric for the formatting and scripts dimensions — heading and fencing conventions, and the agentic-use criteria for bundled scripts |
|
||||
| `references/validation-scripts.md` | Step 1 troubleshooting — the manual structural fallback when `validate.sh` cannot run, and the script exit codes that are easy to misread (loaded only on a script failure) |
|
||||
| `references/sources.md` | Provenance record — agentskills.io sources that informed this skill and which files each contributed to |
|
||||
| `tests/validate.bats` | Bats test suite for validate.sh |
|
||||
| `tests/validate-provenance.bats` | Bats test suite for validate-provenance.sh |
|
||||
| `tests/README.md` | Setup instructions for bats-support and bats-assert test dependencies |
|
||||
| `tests/validate.bats` | (source-only) Bats test suite for validate.sh |
|
||||
| `tests/validate-provenance.bats` | (source-only) Bats test suite for validate-provenance.sh |
|
||||
| `tests/README.md` | (source-only) Setup instructions for bats-support and bats-assert test dependencies |
|
||||
|
||||
Rows marked **(source-only)** exist in the authoring source (`.apm/skills/skill-audit/`) but are
|
||||
not present in an installed plugin: `scripts/sync-plugin-content.sh` strips
|
||||
`<category>/<name>/tests` when it generates the flat mirror, because these are dev-time fixtures no
|
||||
plugin host needs to discover (ADR-0017). Run them from a repo checkout, not from an install.
|
||||
|
||||
@@ -1,19 +1,10 @@
|
||||
---
|
||||
name: skill-audit
|
||||
description: >
|
||||
Use when the user wants to review a skill they wrote, says "audit this skill",
|
||||
"check if my skill follows best practices", "review my SKILL.md", or wants to
|
||||
know if a skill is ready to ship — even if they don't use the word "audit".
|
||||
Also invoke proactively after directly hand-editing a skill's files outside
|
||||
skill-author — an unaudited hand-edit is the same risk as unreviewed code.
|
||||
Audits a skill directory against the agentskills.io specification — structural
|
||||
checks plus qualitative review of description quality, body discipline, patterns,
|
||||
formatting, file structure, scripts, and internal consistency, plus a provenance
|
||||
chain check. Produces a compact findings report
|
||||
(findings only, no PASS noise) with Why and Fix per finding, suitable for agent
|
||||
handoff to /skill-improve or human auditability. Do not use to fix application
|
||||
code bugs or perform general code review unrelated to skill quality.
|
||||
Do not use when the user wants improvements applied — use /skill-improve instead.
|
||||
Use when the user wants a skill directory audited against the agentskills.io
|
||||
spec — "audit this skill", "review my SKILL.md", "is this ready to ship" — or
|
||||
after hand-editing a skill outside skill-author. Not applying fixes ->
|
||||
skill-author.
|
||||
allowed-tools: Bash Read
|
||||
metadata:
|
||||
category: factory
|
||||
@@ -27,9 +18,14 @@ metadata:
|
||||
|
||||
## Gotchas
|
||||
|
||||
- Do not output PASS/FAIL per check while auditing — gather findings internally and surface them only in the Step 4 report. Narrating each check as you go is the default failure mode here.
|
||||
- Do not narrate PASS/FAIL per check while auditing. Gather findings internally and surface them only in the Step 4 report. Narrating each check as you go is the default failure mode here.
|
||||
- A skill carrying `disable-model-invocation: true` is hand-invoked — its description is never routed against, so the trigger, capability and boundary rules do not apply. Audit it as one plain human-facing sentence instead.
|
||||
- `validate.sh` reports two independent length families: the 500-line / 2,770-word pair counts the whole file for spec conformance, while the 250/400-character and 600/900-word pair is the house context budget and its word half counts the **body only**. A skill can sit inside one and fail the other — report them separately.
|
||||
- Vale reporting `0 files` scanned means NOT RUN, not clean. Fall back to full Step 3 judgment for every dimension it would have covered.
|
||||
|
||||
## Step 1 — Structural validation
|
||||
## Step 1 — Deterministic checks
|
||||
|
||||
Resolve all three paths against this skill's own directory so they work from a repo checkout and an installed plugin cache alike. Run exactly:
|
||||
|
||||
```bash
|
||||
bash scripts/validate.sh <skill-dir>
|
||||
@@ -37,98 +33,49 @@ bash scripts/validate-provenance.sh <skill-dir>
|
||||
scripts/vale-wrap.sh <skill-dir>/SKILL.md
|
||||
```
|
||||
|
||||
Note any structural FAILs — they will appear in the report as a `### Structure` dimension. If the script cannot execute (python3 unavailable, Bash denied, or permission error), perform structural checks manually: name format, name matches directory, description length ≤1024 chars, SKILL.md ≤500 lines and ≤2770 words (the word count is a proxy for the ~5,000-token ceiling, and blocks a commit exactly like the line count does), no unfilled `FILL IN:` placeholders, scripts executable and free of interactive prompts.
|
||||
`validate.sh` findings become the `### Structure` dimension — its FAILs and its SUGGESTIONs both.
|
||||
|
||||
Note any Provenance FAILs and INFO findings from `validate-provenance.sh` — they surface in the report as a `### Provenance` dimension (separate from `### Structure`). The script embeds full FAIL/INFO format with Why and Fix per finding; surface them verbatim.
|
||||
If any of the three cannot run, or exits non-zero for a reason other than findings, read `references/validation-scripts.md` — it carries the manual fallback and the misleading exit codes. Ordinary content FAILs are the expected outcome here and need no fallback.
|
||||
|
||||
`vale-wrap.sh` ships inside this skill's own `scripts/` — resolve it relative to this skill's directory the same way `scripts/validate.sh` is resolved above, so the invocation works whether this skill is running from this repo or from an installed plugin cache. Pass no `--config`: handed none, the wrapper loads its own sibling `assets/vale/.vale.ini`, located from the script's path rather than from the cwd. Adding an explicit relative `--config` breaks exactly the case the self-location covers — a resolved script path plus an unresolved config path yields `E100 Runtime error ... does not exist`, exit 2, which the fallback below then misreads as "vale unavailable". It applies that config's `Kyberforge` style — a deterministic prefilter for a subset of the Description/Patterns/Body dimensions below, not a replacement for Step 3. Every Vale alert is a `FAIL` — all rules are graded `error` — so report each one citing its rule ID (e.g. `Kyberforge.DescriptionOpener`). Skip and fall back to Step 3 judgment if the `vale` binary is unavailable. If Vale reports `0 files` scanned, treat the pass as NOT RUN — not as clean — and fall back to full Step 3 judgment for the dimensions it would have covered.
|
||||
`validate-provenance.sh` prints nothing on success. Its FAIL and INFO findings become a separate `### Provenance` dimension, and it emits Why and Fix itself — surface those verbatim.
|
||||
|
||||
## Step 2 — Read all skill files
|
||||
`vale-wrap.sh` applies the bundled `Kyberforge` style as a prefilter. Pass no `--config`; the wrapper locates its own. Every rule is graded `error`, so every alert is a FAIL. Report each one citing its rule ID, filed under the dimension it belongs to, and do not re-derive it by judgment:
|
||||
|
||||
Read every file in the skill directory: `SKILL.md`, `README.md` (if present), all files in `scripts/`, `references/`, `assets/`, and `tests/`. Skip binary files only. Do not skip text files — internal consistency checks require the full picture.
|
||||
| Rule | Dimension |
|
||||
|---|---|
|
||||
| `Kyberforge.DescriptionOpener`, `Kyberforge.CompositionNote`, `Kyberforge.VagueWording` | description |
|
||||
| `Kyberforge.SentenceOpenerThereIs` | body-discipline |
|
||||
| `Kyberforge.PaddingPhrase` | patterns |
|
||||
|
||||
## Step 2 — Read the whole skill
|
||||
|
||||
Read `SKILL.md`, `README.md`, and every text file under `scripts/`, `references/`, `assets/` and `tests/`. Skip binaries only — internal-consistency findings need the full picture.
|
||||
|
||||
## Step 3 — Qualitative audit
|
||||
|
||||
Work through each dimension internally. Collect findings only; report them in Step 4. Cite file and line number for every finding.
|
||||
Load a dimension's rubric before judging that dimension. Each is self-contained, and each is grounded in the agentskills.io specification plus the house context-budget contract (ADR-0020).
|
||||
|
||||
### Description
|
||||
| Dimension | Read |
|
||||
|---|---|
|
||||
| description | `references/description-quality.md` |
|
||||
| body-discipline | `references/body-discipline.md` |
|
||||
| patterns | `references/patterns.md` |
|
||||
| file-structure, internal-consistency | `references/file-structure.md` |
|
||||
| formatting, scripts | `references/formatting-and-scripts.md` |
|
||||
|
||||
Vale's `Kyberforge.DescriptionOpener` ("This skill..." openers) and `Kyberforge.VagueWording` (filler like "helps with", "utilize") alerts from Step 1 — both FAILs — cover imperative phrasing and known vague-wording filler directly; report them as findings without re-deriving by judgment. The rest is still a judgment call:
|
||||
|
||||
- **Action-verb opening**: does the description start with a verb ("Audits...", "Reviews...", "Validates...")? Vale's `Kyberforge.DescriptionOpener` alert only catches the literal "This skill..." pattern — confirming an arbitrary opening word is genuinely a strong verb still requires judgment.
|
||||
- **Specificity beyond the filler blocklist**: are capabilities stated precisely ("parses OpenAPI specs") or genuinely vaguely ("handles files")?
|
||||
- **Indirect triggers**: does it cover cases where the user doesn't name the domain directly?
|
||||
- **Near-miss exclusions**: are "Do not use when..." clauses present if a near-miss skill could steal activations?
|
||||
- **Length**: under 1024 characters?
|
||||
|
||||
If a description finding is borderline or the distinction between PASS and FAIL is unclear, read `references/description-quality.md`.
|
||||
|
||||
### Body discipline
|
||||
|
||||
For each sentence in the body, apply: *"Would the agent get this wrong without this sentence?"* Flag any that answer "no" as padding.
|
||||
|
||||
- **Defaults not menus**: every decision point gives one default + one escape hatch, not a list of options
|
||||
- **Why rationale**: include/exclude rules explain why, not just what
|
||||
- **Control calibration**: prescriptive for fragile or critical sequences (e.g. a script invocation where flag order or exact arguments must not change); flexible where multiple approaches are valid
|
||||
|
||||
Vale's `Kyberforge.SentenceOpenerThereIs` alert from Step 1 (FAIL — sentences starting with "There is"/"There are") covers pattern-matchable body-wide filler directly; report it as a finding without re-deriving by judgment.
|
||||
|
||||
If uncertain whether a sentence is padding or whether a control decision is correctly calibrated, read `references/body-discipline.md`.
|
||||
|
||||
### Patterns
|
||||
|
||||
Check each pattern is appropriate and correctly formed:
|
||||
|
||||
- **Gotchas**: placed near the top; each entry is a specific fact that defies a reasonable assumption — not a general tip
|
||||
- **Prescriptive sequence**: inner code fences escaped as `\`\`\`` when nested inside a markdown block
|
||||
- **Checklists**: used for multi-step workflows, not single steps
|
||||
- **Conditional references**: specific trigger stated ("If X, read `references/file.md`") — not a generic "see references/". Vale's `Kyberforge.PaddingPhrase` alert from Step 1 flags the generic phrasing directly; other malformed conditional-reference forms still require judgment.
|
||||
- **Output templates**: present when the agent must produce a specific format; absent otherwise
|
||||
|
||||
### File structure
|
||||
|
||||
- Permitted directories: `scripts/`, `references/`, `assets/`, `tests/`; flag any other unlisted directory as FAIL — the spec allows additional dirs but this skill permits only these four to keep skills focused
|
||||
- `scripts/` contains only executable code agents can run; test files (`.bats`, `*_test.*`, `test_*.sh`) in `scripts/` are a FAIL — they belong in `tests/`
|
||||
- No non-spec files at the skill root (e.g. META.md, extra config files outside permitted directories)
|
||||
- Optional directories contain real content — not just unfilled placeholder READMEs
|
||||
- `README.md` present and accurately describes the skill and its files
|
||||
- No cross-plugin path references in SKILL.md, scripts/, references/, or assets/ — paths using `../`, `../../`, or absolute repo paths (e.g. `plugins/<plugin>/skills/<other-skill>/`, or its APM-native equivalent `.apm/skills/<other-skill>/`) break when the plugin is installed to a cache; flag any found
|
||||
- `references/sources.md` is exempt from the cross-plugin path check — `Research doc:` fields are development-only provenance pointers, not runtime references; they intentionally reference paths outside the skill directory and are expected to be non-resolvable after plugin install; `validate-provenance.sh` handles this gracefully by silently skipping upstream checks when those paths don't resolve
|
||||
- `tests/` is exempt from the cross-plugin path check — test files are dev-only and may reference repo-level test infrastructure (e.g. a shared `tests/test_helper/`). This dependency must be declared in `tests/README.md`; flag if tests exist but `tests/README.md` is absent or does not document the dependency
|
||||
|
||||
### Formatting
|
||||
|
||||
- Heading levels consistent: H2 for main sections, H3 for subsections
|
||||
- Code blocks fenced with a language tag where applicable (`bash`, `markdown`, `python`)
|
||||
- Consistent whitespace: blank line between sections, consistent list indentation
|
||||
- No broken relative paths in file references
|
||||
|
||||
### Scripts
|
||||
|
||||
- No interactive TTY prompts (`read`, `input()`, `readline`)
|
||||
- `--help` exposed with concise usage
|
||||
- Data to stdout, diagnostics to stderr
|
||||
- Idempotent ("create if not exists")
|
||||
- Meaningful exit codes documented in `--help`
|
||||
- `--dry-run` present for destructive operations
|
||||
|
||||
### Internal consistency
|
||||
|
||||
- SKILL.md steps match what scripts actually do
|
||||
- `README.md` file table lists every file that exists — no missing entries, no stale entries
|
||||
- Placeholder READMEs in `scripts/`, `references/`, `assets/` consistent with what SKILL.md says about each directory
|
||||
Cite file and line number for every finding.
|
||||
|
||||
## Step 4 — Report
|
||||
|
||||
Open with a coverage line listing every dimension checked:
|
||||
Open with a coverage line naming every dimension checked:
|
||||
|
||||
```text
|
||||
Checked: structure · description · body-discipline · patterns · file-structure · formatting · scripts · internal-consistency · provenance
|
||||
```
|
||||
|
||||
Then output only dimensions that have findings, grouped under H3 headings, FAILs before SUGGESTIONs within each dimension. Omit clean dimensions entirely — their absence confirms they passed.
|
||||
Then output only the dimensions that have findings, grouped under H3 headings, FAILs before SUGGESTIONs within each. Omit clean dimensions — their absence is what confirms they passed.
|
||||
|
||||
For each finding:
|
||||
Each finding:
|
||||
|
||||
```text
|
||||
FAIL/SUGGESTION <finding> — file:line
|
||||
@@ -136,18 +83,4 @@ FAIL/SUGGESTION <finding> — file:line
|
||||
Fix: <exact change — quote before/after where applicable>
|
||||
```
|
||||
|
||||
Close with a result block:
|
||||
|
||||
```text
|
||||
## Result
|
||||
|
||||
PASS
|
||||
PASS (N suggestions)
|
||||
PASS · P info
|
||||
PASS (N suggestions) · P info
|
||||
FAIL (N fails · M suggestions)
|
||||
FAIL (N fails · M suggestions) · P info
|
||||
Run /skill-improve to address findings.
|
||||
```
|
||||
|
||||
INFO findings are observational — do not affect PASS/FAIL. Omit `· P info` when there are no INFO findings. Omit the `/skill-improve` line when there are no findings at all. Do not apply fixes — report and propose only.
|
||||
Close with a `## Result` block holding one line: `PASS`, `PASS (N suggestions)`, or `FAIL (N fails · M suggestions)`, each optionally followed by ` · P info`. INFO findings are observational and never change PASS/FAIL; omit `· P info` when there are none. Add a second line, `Run skill-author to address findings.`, whenever there is at least one finding. Do not apply fixes — report and propose only.
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
extends: existence
|
||||
message: "Composition or architecture note in a description: '%s' — a description carries a trigger, one capability clause and a boundary clause only; move this to README.md"
|
||||
level: error
|
||||
scope: text.frontmatter.description
|
||||
ignorecase: true
|
||||
tokens:
|
||||
- cross-cutting
|
||||
- shared (skill|agent)
|
||||
- human-facing
|
||||
- entry[- ]point
|
||||
- composes
|
||||
- rather than duplicating
|
||||
- replaces the (old|former|previous)
|
||||
@@ -4,4 +4,4 @@ level: error
|
||||
scope: text.frontmatter.description
|
||||
ignorecase: true
|
||||
raw:
|
||||
- '^This (skill|agent)\b'
|
||||
- '^This\b'
|
||||
|
||||
@@ -6,31 +6,123 @@ source_keys:
|
||||
|
||||
# Body Discipline Reference
|
||||
|
||||
Source: agentskills.io — skill-authoring
|
||||
Upstream source: agentskills.io — skill-authoring, best-practices.
|
||||
House contract: ADR-0020, the context budget.
|
||||
|
||||
## The core test
|
||||
|
||||
For every sentence in the body, ask: **"Would the agent get this wrong without this instruction?"**
|
||||
|
||||
If no — cut it. The agent already knows it from general training. Adding it wastes tokens and dilutes the signal of what matters.
|
||||
If no — cut it. The agent already knows it from general training. Adding it wastes tokens and
|
||||
dilutes the signal of what matters.
|
||||
|
||||
## What belongs in the body
|
||||
## What the body is for
|
||||
|
||||
The body carries the **decision procedure only**: ordered steps, decision branches, gates, and
|
||||
which reference to load when.
|
||||
|
||||
Include content the agent lacks:
|
||||
|
||||
- Project-specific conventions and domain procedures it cannot infer
|
||||
- Non-obvious edge cases and environment-specific gotchas
|
||||
- The specific tools or sequences to use (not the full range of options)
|
||||
- The specific tools or sequences to use — not the full range of options
|
||||
- One default per decision point with one escape hatch
|
||||
|
||||
Do not include:
|
||||
Move to `references/`, behind an explicit "If X, read `references/<file>.md`" trigger — the literal
|
||||
conditional form, never a generic pointer. Write the real filename in the skill under audit; the
|
||||
angle brackets are a placeholder here, and a literal `references/file.md` in a body is an ERROR
|
||||
from the ADR-0020 gate because no such file exists on disk. Move:
|
||||
|
||||
- Lookup tables and spec restatements
|
||||
- Output schemas, templates and example blocks
|
||||
- Rationale and justification prose
|
||||
- Anything only one branch of the procedure ever reaches
|
||||
|
||||
Do not include at all:
|
||||
|
||||
- Concepts the agent already knows (what JSON is, how HTTP works, what a CSV is)
|
||||
- Exhaustive option lists — pick a default; the agent doesn't benefit from choosing
|
||||
- Exhaustive option lists — pick a default; the agent does not benefit from choosing
|
||||
- Steps the agent handles independently — over-specifying leads to unproductive paths
|
||||
- Restatements of the description — it's already in context
|
||||
- Restatements of the description, which is already in context
|
||||
|
||||
## Two length families, measured differently
|
||||
|
||||
Do not conflate these, and do not report them as one finding.
|
||||
|
||||
| Gate | SUGGESTION | FAIL | Counts |
|
||||
|---|---|---|---|
|
||||
| Body budget (house, ADR-0020) | 600 words | 900 words | the **body only** — everything after the frontmatter's closing `---` |
|
||||
| Spec conformance (agentskills.io) | — | 2,770 words / 500 lines | the **whole file**, frontmatter included |
|
||||
|
||||
The 2,770-word ceiling is a token-conformance backstop calibrated to the densest prose in the
|
||||
corpus; it says nothing about quality and a file can sit a thousand words inside it while failing
|
||||
the body budget. The 900-word ceiling is the quality gate: a body is loaded into the caller's live
|
||||
context and competes with the conversation already there. `validate.sh` reports both. Cite whichever
|
||||
one actually fired.
|
||||
|
||||
A word count cannot detect the defect it stands in for. Treat both numbers as backstops to the
|
||||
dispatch rule and the Gotchas constraint below, never as a substitute for them.
|
||||
|
||||
## Dispatch is mandatory at two or more mutually exclusive flows
|
||||
|
||||
If a skill handles two or more flows that a single invocation cannot both take — separate
|
||||
subcommands, separate input types, separate lifecycle stages — the body carries a **dispatch
|
||||
table** plus the gates common to every branch, and each flow lives in its own self-contained
|
||||
`references/` file. Inlining all of them is a FAIL regardless of word count, because every
|
||||
invocation then pays for every branch it did not take.
|
||||
|
||||
The reference shape in this repo is `apm-workflow`: a **421-word body** dispatching to roughly
|
||||
3,000 words of references across five mutually exclusive invocations. Its whole-file count is 554
|
||||
words — cite 421 when calibrating a body, or the conflation this section warns against reappears
|
||||
in the finding itself.
|
||||
|
||||
## Gotchas sections
|
||||
|
||||
The highest-value construct in a body, and the easiest to fill with noise. A Gotcha must state a
|
||||
fact that **contradicts a reasonable default** — something the agent gets wrong precisely by acting
|
||||
sensibly.
|
||||
|
||||
```markdown
|
||||
## Gotchas
|
||||
- The `users` table uses soft deletes. Always include `WHERE deleted_at IS NULL`.
|
||||
- User ID is `user_id` in the database, `uid` in auth, `accountId` in billing. Same value.
|
||||
```
|
||||
|
||||
Constraints:
|
||||
|
||||
- **More than five entries is a SUGGESTION** — five is the guideline, not a ceiling. Past five, the
|
||||
section is usually a summary of the body rather than a set of traps, and the agent stops reading
|
||||
it as a warning. It stays advisory because whether a given gotcha earns its place is judgment;
|
||||
`validate.sh` emits it through `suggest()` and the run still exits 0.
|
||||
- **A Gotcha that paraphrases a step in the body below it is a FAIL.** It has no independent
|
||||
content, and it teaches the agent that Gotchas can be skimmed because the real instruction is
|
||||
coming. This one is the auditor's call — no script detects it.
|
||||
- **A Gotchas section exceeding 25% of the body is a SUGGESTION** — the body has been inverted into
|
||||
a preamble. Same tier and same reasoning as the entry count, and independent of it: either can
|
||||
fire without the other.
|
||||
- Place the section near the top. A gotcha read after the mistake is worthless, which is also why
|
||||
Gotchas is the one construct exempt from moving to `references/`.
|
||||
|
||||
Worked negative example — `git-commits` carries twelve entries, of which four restate content
|
||||
that already appears below or in the description:
|
||||
|
||||
| Gotcha | Restates |
|
||||
|---|---|
|
||||
| `:31` "Communicates SemVer impact" | the description |
|
||||
| `:32` "Confirmation gates are mandatory for destructive operations" | step 9 at `:52` |
|
||||
| `:33` "Never skip hooks with `--no-verify`" | step 9 at `:52` |
|
||||
| `:36` "Never commit secrets" | step 2 at `:45` |
|
||||
|
||||
All four are FAILs under the paraphrase rule. The entry count and the section's share of the body
|
||||
(387 of 1,102 words, 35%) are two further SUGGESTIONs on top — the script reports both, and neither
|
||||
fails the run on its own. What makes this worth auditing directly is that the four paraphrase FAILs
|
||||
pass every word gate there is; only reading the construct finds them.
|
||||
|
||||
## Calibrating control
|
||||
|
||||
**Be prescriptive** when operations are fragile, consistency matters, or a specific sequence must be followed:
|
||||
**Be prescriptive** when operations are fragile, consistency matters, or a specific sequence must be
|
||||
followed:
|
||||
|
||||
```markdown
|
||||
Run exactly:
|
||||
\`\`\`bash
|
||||
@@ -39,11 +131,13 @@ python scripts/migrate.py --verify --backup
|
||||
Do not modify the command or add additional flags.
|
||||
```
|
||||
|
||||
**Give freedom** when multiple approaches are valid. Explaining *why* outperforms rigid directives — agents make better decisions when they understand the purpose.
|
||||
**Give freedom** when multiple approaches are valid. Explaining *why* outperforms rigid directives —
|
||||
agents make better decisions when they understand the purpose.
|
||||
|
||||
## Defaults not menus
|
||||
|
||||
Never present a list of equivalent options — pick one and mention the alternative briefly:
|
||||
|
||||
```markdown
|
||||
# Too many options
|
||||
Use pypdf, pdfplumber, PyMuPDF, or pdf2image...
|
||||
@@ -52,37 +146,23 @@ Use pypdf, pdfplumber, PyMuPDF, or pdf2image...
|
||||
Use pdfplumber for text extraction. For scanned PDFs requiring OCR, use pdf2image instead.
|
||||
```
|
||||
|
||||
## Gotchas sections
|
||||
|
||||
Highest value content — environment-specific facts that defy reasonable assumptions. Place near the top of the body so the agent reads them before encountering the situation.
|
||||
|
||||
```markdown
|
||||
## Gotchas
|
||||
- The `users` table uses soft deletes. Always include `WHERE deleted_at IS NULL`.
|
||||
- User ID is `user_id` in the database, `uid` in auth, `accountId` in billing. Same value.
|
||||
```
|
||||
|
||||
Each entry must be a specific, surprising fact — not a general tip or reminder.
|
||||
|
||||
## Progressive disclosure
|
||||
|
||||
Keep `SKILL.md` under 500 lines. When more content is needed, move it to `references/` and load conditionally:
|
||||
|
||||
```markdown
|
||||
If the API returns a non-200 status, read `references/api-errors.md`.
|
||||
```
|
||||
|
||||
"If X, read Y" is more useful than "see references/ for details." The agent loads on demand rather than up front.
|
||||
|
||||
## Auditing guidance
|
||||
|
||||
Flag as FAIL if:
|
||||
- A sentence answers "no" to the core test (would agent get this wrong without it?) — it is padding
|
||||
- Decision points present a menu of options with no default
|
||||
- Instructions repeat content already in the description
|
||||
- Prescriptive sequences are used where flexibility is fine, or vice versa
|
||||
|
||||
- A sentence answers "no" to the core test — it is padding
|
||||
- The body exceeds 900 words counted body-only (`validate.sh` reports it)
|
||||
- Two or more mutually exclusive flows are inlined instead of dispatched
|
||||
- A Gotcha paraphrases a step in the body below it
|
||||
- A decision point presents a menu of options with no default
|
||||
- An instruction repeats content already in the description
|
||||
- A prescriptive sequence is used where flexibility is fine, or the reverse
|
||||
|
||||
Flag as SUGGESTION if:
|
||||
- A rationale is missing from an include/exclude rule (present but unexplained)
|
||||
|
||||
- The body exceeds 600 words counted body-only but stays at or under 900
|
||||
- The Gotchas section carries more than five entries
|
||||
- The Gotchas section exceeds 25% of the body
|
||||
- A rationale is missing from an include/exclude rule — present but unexplained
|
||||
- Gotchas are correct but placed late in the body rather than near the top
|
||||
- A conditional reference trigger is vague ("see references/") rather than specific ("If X, read Y")
|
||||
- Content that only one branch reaches is inlined where a `references/` file would serve
|
||||
|
||||
@@ -6,49 +6,112 @@ source_keys:
|
||||
|
||||
# Description Quality Reference
|
||||
|
||||
Source: agentskills.io — optimizing-descriptions
|
||||
Upstream source: agentskills.io — optimizing-descriptions, specification.
|
||||
House contract: ADR-0020, the context budget. The house contract is narrower than the spec
|
||||
rather than a reinterpretation of it: where both speak, both must be satisfied.
|
||||
|
||||
## How triggering works
|
||||
## Why the description is the expensive part
|
||||
|
||||
At startup, agents load only the `name` and `description` of each skill. When a user's task matches a description, the agent reads the full `SKILL.md` into context. **The description carries the entire triggering burden** — the body is never seen until after triggering.
|
||||
At startup an agent loads only the `name` and `description` of every installed skill. The body is
|
||||
never seen until the skill triggers. The description therefore carries the entire triggering
|
||||
burden **and** is paid for in every session, whether the skill fires or not.
|
||||
|
||||
Agents typically consult skills only for tasks requiring knowledge beyond their defaults. Specialized knowledge — unfamiliar APIs, domain-specific workflows, uncommon formats — is where description wording makes the difference.
|
||||
A second cost is less obvious and is a correctness hazard rather than a token cost: a description
|
||||
that summarises the workflow is a shortcut the agent takes *instead of* reading the body. A
|
||||
measured failure upstream — a description saying "code review between tasks" — produced one review
|
||||
where the body's flowchart specified two.
|
||||
|
||||
## What a good description does
|
||||
## Step 0 — establish which contract applies
|
||||
|
||||
- **Imperative phrasing** — "Use when..." not "This skill does...". The agent is deciding whether to act.
|
||||
- **User intent, not mechanics** — describe what the user is trying to achieve, not how the skill works internally.
|
||||
- **Err toward being pushy** — explicitly name contexts where the skill applies, including cases where the user doesn't name the domain: "even if they don't mention X explicitly."
|
||||
- **Specificity over vagueness** — "parses and validates OpenAPI specs" beats "helps with APIs."
|
||||
- **Near-miss exclusions** — add "Do not use when..." only if a near-miss skill exists that could steal activations. Use strong near-misses (queries that share keywords but need something different), not weak ones ("write a fibonacci function").
|
||||
- **Hard limit: 1024 characters** — descriptions grow during revision; check length before finalising.
|
||||
Read the frontmatter before judging a single word.
|
||||
|
||||
- **`disable-model-invocation: true`** — the skill is hand-invoked. Its description is never
|
||||
matched against user intent, so it is not a routing string. It carries **one plain human-facing
|
||||
sentence** stating what the skill does. Audit it for that and nothing else. Reporting a missing
|
||||
trigger clause, a missing boundary clause or absent indirect triggers on a hand-invoked skill is
|
||||
a wrong finding, not a strict one.
|
||||
- **No such flag** — the skill is model-invoked and the rest of this file applies.
|
||||
|
||||
## The three-part shape
|
||||
|
||||
A model-invoked description carries exactly three things:
|
||||
|
||||
1. **Trigger clause.** When to invoke, phrased imperatively: `Use when ...`. Not `This skill ...` —
|
||||
the agent is deciding whether to act, not reading a catalogue entry.
|
||||
2. **At most one capability clause.** What it does, in one clause. Never an enumeration.
|
||||
3. **Boundary clause.** Compressed form: `Not <thing> -> <skill-name>.` The target must resolve to
|
||||
a real skill directory or agent file in the authoring source; `validate.sh` checks that
|
||||
deterministically and a dangling target already surfaces as a Structure FAIL.
|
||||
|
||||
Everything else belongs in the body or in `README.md`.
|
||||
|
||||
## Indirect triggers — conditional, never blanket
|
||||
|
||||
Add "even if the user doesn't say X" **only where the user's natural phrasing genuinely omits the
|
||||
domain word.** True for the `gitea-*` family: people say "create an issue", not "create a Gitea
|
||||
issue". False for `git-commits`: nobody asks for a commit without saying commit. A blanket
|
||||
indirect-trigger clause on a skill whose domain word is unavoidable is padding charged to every
|
||||
session.
|
||||
|
||||
## Near-miss exclusions
|
||||
|
||||
Add a boundary clause only where a sibling skill could plausibly steal the activation. Use strong
|
||||
near-misses — queries that share keywords but need something different — not weak ones ("write a
|
||||
fibonacci function"). One boundary clause per genuine near-miss; a list of four is enumeration
|
||||
wearing a boundary's clothes.
|
||||
|
||||
## Before / after
|
||||
|
||||
```yaml
|
||||
# Weak
|
||||
description: Process CSV files.
|
||||
|
||||
# Strong
|
||||
# FAIL — enumeration first, mechanics as the opener, a blanket indirect trigger,
|
||||
# and 300+ characters of it preloaded into every session forever.
|
||||
description: >
|
||||
Analyze CSV and tabular data files — compute summary statistics,
|
||||
add derived columns, generate charts, and clean messy data. Use when
|
||||
the user has a CSV, TSV, or Excel file and wants to explore, transform,
|
||||
or visualize the data, even if they don't explicitly mention "CSV" or
|
||||
"analysis."
|
||||
Analyze CSV and tabular data files — compute summary statistics, add derived
|
||||
columns, generate charts, and clean messy data. Use when the user has a CSV,
|
||||
TSV, or Excel file and wants to explore, transform, or visualize the data,
|
||||
even if they don't explicitly mention "CSV" or "analysis."
|
||||
|
||||
# PASS — trigger, one capability clause, boundary. The four verbs the FAIL
|
||||
# version enumerates are the body's job; the router cannot act on them.
|
||||
description: >
|
||||
Use when the user has a CSV, TSV, or Excel file and wants it explored,
|
||||
transformed, or charted. Not schema design -> data-model.
|
||||
```
|
||||
|
||||
The strong version names capabilities precisely and broadens applicability beyond explicit keyword matches.
|
||||
(`data-model` is illustrative. In a real description the target has to resolve.)
|
||||
|
||||
## Auditing guidance
|
||||
|
||||
Flag as FAIL if:
|
||||
- Phrasing is descriptive ("This skill...") not imperative ("Use when...")
|
||||
- Capabilities are vague ("helps with APIs") — require precise verbs and nouns
|
||||
- No indirect trigger coverage when indirect cases clearly exist
|
||||
- No near-miss exclusions when a sibling skill could plausibly steal activations
|
||||
- Length exceeds 1024 characters
|
||||
|
||||
- **Over 400 characters.** Measured on the folded YAML value, not the raw source lines.
|
||||
`validate.sh` reports the number; do not re-derive it, but do point the Fix at what to cut.
|
||||
- **Internal mechanics appear in the description.** Any of:
|
||||
- capability enumeration or a feature list;
|
||||
- output-format detail ("Produces a compact findings report with Why and Fix per finding");
|
||||
- composition or architecture notes ("composes X rather than duplicating Y", "a cross-cutting
|
||||
shared skill", "the human-facing entry point", "replaces the old flat invocation");
|
||||
- implementation detail ("self-validates via a bundled deterministic script").
|
||||
|
||||
None of it can change a routing decision and all of it is preloaded.
|
||||
`Kyberforge.CompositionNote` catches the common phrasings deterministically; the rest is
|
||||
judgment. This is the rule that deflates a description, so apply it before reaching for length.
|
||||
- **The same trigger stated twice in two registers** — a verb list, then the same verbs re-quoted
|
||||
as user phrasings, usually in the same order. One register, whichever routes better.
|
||||
- **Descriptive rather than imperative phrasing** (`This skill ...`, `This is the ...`).
|
||||
`Kyberforge.DescriptionOpener` catches any opener matching `^This`.
|
||||
- **Vague capabilities** ("helps with APIs" where "parses and validates OpenAPI specs" was
|
||||
available). `Kyberforge.VagueWording` catches the known filler; imprecision outside that list is
|
||||
judgment.
|
||||
- **A boundary clause naming a target that does not resolve** to a real skill directory or agent
|
||||
file in the authoring source. `validate.sh` reports the unresolved name.
|
||||
- **Trigger-list, boundary or indirect-trigger content on a hand-invoked skill** — see Step 0.
|
||||
- **Over 1024 characters** — the agentskills.io specification ceiling, unchanged and independent
|
||||
of the 400-character house ceiling above.
|
||||
|
||||
Flag as SUGGESTION if:
|
||||
- Indirect trigger coverage exists but could be more specific
|
||||
- Near-miss exclusions are present but target weak near-misses only
|
||||
|
||||
- **Over 250 characters** but at or under 400. This tier is what moves the corpus average; the FAIL
|
||||
tier only stops outliers. Report it rather than treating a 399-character description as clean.
|
||||
- A near-miss exclusion is present but targets a weak near-miss.
|
||||
- An indirect trigger is present and warranted but could name the omitted phrasing more precisely.
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
---
|
||||
source_keys:
|
||||
- agentskills-spec
|
||||
---
|
||||
|
||||
# File Structure and Internal Consistency Reference
|
||||
|
||||
Upstream source: agentskills.io — specification (optional directories, file references).
|
||||
|
||||
Read this when judging the **file-structure** and **internal-consistency** dimensions.
|
||||
|
||||
## Permitted directories
|
||||
|
||||
Only four: `scripts/`, `references/`, `assets/`, `tests/`. The specification permits additional
|
||||
directories; this house does not, because an unlisted directory is content no auditor and no host
|
||||
knows to look at. Flag any other directory as a FAIL.
|
||||
|
||||
- `scripts/` holds only executable code an agent can run. Test files (`.bats`, `*_test.*`,
|
||||
`test_*.sh`) there are a FAIL — they belong in `tests/`.
|
||||
- No non-spec files at the skill root: no `META.md`, no stray config outside the four directories.
|
||||
- An optional directory that exists must hold real content, not an unfilled placeholder README.
|
||||
- `README.md` is present and describes the skill and its files accurately.
|
||||
|
||||
## Cross-plugin path references
|
||||
|
||||
A plugin is copied to a cache on install, and a path that climbs out of the skill directory stops
|
||||
resolving there. Flag any `../`, `../../`, or absolute repo path (`plugins/<plugin>/skills/<other>/`
|
||||
and its APM-native equivalent `.apm/skills/<other>/`) appearing in `SKILL.md`, `scripts/`,
|
||||
`references/` or `assets/`.
|
||||
|
||||
**Referring to another skill's file.** There is one sanctioned spelling, and it is possessive:
|
||||
`skill-audit's references/validation-scripts.md`. Write the skill by name and let the reader
|
||||
resolve it — do not spell the repo path. The full path is the thing this section forbids, and
|
||||
`references/validation-scripts.md` on its own is a hard ERROR from the ADR-0020 gate, which
|
||||
requires an unqualified `references/` pointer to exist in the skill's OWN directory. The
|
||||
possessive form is the only spelling both rules accept; the gate recognises it and skips the
|
||||
on-disk check. Flag any other spelling of a cross-skill reference.
|
||||
|
||||
Two directories are exempt, and the exemptions are structural rather than discretionary:
|
||||
|
||||
- **`references/sources.md`.** Its `Research doc:` fields are development-time provenance pointers,
|
||||
not runtime references. They are expected to be unresolvable after install, and
|
||||
`validate-provenance.sh` handles that by skipping upstream checks silently when the path is
|
||||
absent. Flagging them would make every correctly-provenanced skill fail.
|
||||
- **`tests/`.** Test files are dev-only and may reference repo-level infrastructure such as a shared
|
||||
`tests/test_helper/`. The exemption is conditional on the dependency being declared: if `tests/`
|
||||
exists and `tests/README.md` is absent or does not document it, that is a FAIL.
|
||||
|
||||
## Internal consistency
|
||||
|
||||
The skill has to agree with itself. Three checks:
|
||||
|
||||
- `SKILL.md`'s steps match what the scripts actually do — the arguments, the exit codes, and the
|
||||
output shape it tells the agent to expect.
|
||||
- `README.md`'s file table lists every file that exists, with no missing rows and no stale rows for
|
||||
files since deleted.
|
||||
- Placeholder READMEs inside `scripts/`, `references/` and `assets/` say the same thing about each
|
||||
directory that `SKILL.md` does.
|
||||
|
||||
A stale README row is the most common finding here and the easiest to miss from inside an
|
||||
authoring pass, because the author knows what was intended and reads it into the gap.
|
||||
|
||||
## Auditing guidance
|
||||
|
||||
Flag as FAIL if:
|
||||
|
||||
- A directory outside the four permitted ones exists
|
||||
- Test files sit in `scripts/`
|
||||
- A non-spec file sits at the skill root
|
||||
- A cross-plugin or parent-relative path appears outside the two exempt locations
|
||||
- `tests/` exists but `tests/README.md` is missing or does not document its repo-level dependency
|
||||
- `README.md` is absent, or its file table has a missing or stale row
|
||||
- `SKILL.md` describes a script invocation the script does not accept
|
||||
|
||||
Flag as SUGGESTION if:
|
||||
|
||||
- An optional directory exists but holds only a placeholder README
|
||||
- `README.md` is accurate but describes a file's purpose more thinly than `SKILL.md` does
|
||||
@@ -0,0 +1,63 @@
|
||||
---
|
||||
source_keys:
|
||||
- agentskills-spec
|
||||
- agentskills-using-scripts
|
||||
---
|
||||
|
||||
# Formatting and Scripts Reference
|
||||
|
||||
Upstream source: agentskills.io — specification (body content), using-scripts (designing scripts
|
||||
for agentic use).
|
||||
|
||||
Read this when judging the **formatting** and **scripts** dimensions. Both are checklists of static
|
||||
criteria that never vary by skill, which is exactly why they live here rather than in the body.
|
||||
|
||||
## Formatting
|
||||
|
||||
- Heading levels are consistent: H2 for main sections, H3 for subsections. A body that jumps from
|
||||
H2 to H4, or opens on H3, reads as a fragment of a larger document.
|
||||
- Code blocks carry a language tag wherever one applies — `bash`, `markdown`, `python`, `yaml`,
|
||||
`text`. An untagged block loses syntax highlighting and, more importantly, loses the signal of
|
||||
what the agent is meant to do with it.
|
||||
- Whitespace is consistent: a blank line between sections, one list-indentation style throughout.
|
||||
- No broken relative paths in file references. Every `references/…`, `scripts/…` and `assets/…`
|
||||
path named in the body resolves against the skill directory.
|
||||
|
||||
## Scripts
|
||||
|
||||
A script in a skill is run by an agent with no terminal and no human to answer it. The criteria
|
||||
follow from that:
|
||||
|
||||
- **No interactive TTY prompts** — no `read`, no `input()`, no `readline`. A script that blocks on
|
||||
a prompt hangs the run with no diagnostic. `validate.sh` detects the common forms and reports
|
||||
them under Structure; the judgment call is any prompt it cannot pattern-match. What counts is
|
||||
where stdin comes from, not the word `read`: a `read` fed by a here-string, a here-doc, a pipe,
|
||||
or a redirect from a file never touches a terminal and is not a finding. `validate.sh` excludes
|
||||
those forms, so do not rewrite a working `read -r A B <<< "$line"` into parameter expansion to
|
||||
satisfy this rule.
|
||||
- **`--help` is exposed** and gives concise usage.
|
||||
- **Data to stdout, diagnostics to stderr.** A caller piping the script has to be able to separate
|
||||
the result from the commentary.
|
||||
- **Idempotent** — "create if not exists" rather than "create", so a re-run after a partial failure
|
||||
is safe.
|
||||
- **Meaningful exit codes, documented in `--help`.** An agent branches on the exit code; an
|
||||
undocumented one is a coin flip.
|
||||
- **`--dry-run` present for destructive operations.**
|
||||
|
||||
## Auditing guidance
|
||||
|
||||
Flag as FAIL if:
|
||||
|
||||
- A script prompts interactively, in any form
|
||||
- A script exposes no `--help`
|
||||
- A destructive script has no `--dry-run`
|
||||
- Data and diagnostics share a stream, so the output cannot be piped
|
||||
- A relative path named in the body does not resolve
|
||||
- Heading levels are inconsistent enough to break the document's structure
|
||||
|
||||
Flag as SUGGESTION if:
|
||||
|
||||
- Exit codes are meaningful but undocumented in `--help`
|
||||
- A code block is untagged where a language applies
|
||||
- A script is idempotent in practice but does not say so, leaving a re-run's safety unclear
|
||||
- List indentation or section spacing is inconsistent without breaking the render
|
||||
@@ -0,0 +1,68 @@
|
||||
---
|
||||
source_keys:
|
||||
- agentskills-spec
|
||||
- agentskills-best-practices
|
||||
---
|
||||
|
||||
# Patterns Reference
|
||||
|
||||
Upstream source: agentskills.io — best-practices (instruction patterns), specification.
|
||||
|
||||
Read this when judging the **patterns** dimension: whether each instruction construct a skill uses
|
||||
is the right construct for the job and is correctly formed. Formation, not content — a Gotcha's
|
||||
*content* is judged in `references/body-discipline.md`.
|
||||
|
||||
## The constructs and when each is right
|
||||
|
||||
| Construct | Right when | Wrong when |
|
||||
|---|---|---|
|
||||
| Gotchas | An environment fact contradicts a reasonable default | Used as a summary of the steps below |
|
||||
| Prescriptive sequence | The operation is fragile and flag order or exact arguments must not change | Several approaches are equally valid |
|
||||
| Checklist | A multi-step workflow the agent must complete in order | A single step dressed up as a list |
|
||||
| Conditional reference | Detail is needed on one branch only | The reference is needed on every run and is loaded blind |
|
||||
| Output template | The agent must emit a specific format a caller consumes | The output is prose nobody parses |
|
||||
|
||||
## Formation rules
|
||||
|
||||
**Gotchas** sit near the top of the body, before the steps that would otherwise walk into them.
|
||||
Placement late in the body is a SUGGESTION, not a FAIL — the content is still correct, it is just
|
||||
read after the mistake.
|
||||
|
||||
**Prescriptive sequences** that quote a fenced block inside another markdown block must escape the
|
||||
inner fence as `` \`\`\` ``. An unescaped inner fence terminates the outer block and the remaining
|
||||
instructions render as prose.
|
||||
|
||||
**Conditional references** state a specific trigger, naming a file that exists in the skill's own
|
||||
`references/` directory:
|
||||
|
||||
```text
|
||||
If the API returns a non-200 status, read `references/api-errors.md`.
|
||||
```
|
||||
|
||||
That block is fenced because the filename in it is illustrative — an unfenced `references/` pointer
|
||||
in a `SKILL.md` body must resolve on disk or the ADR-0020 gate reports a hard ERROR. The generic
|
||||
form — pointing at the directory and hoping — defeats
|
||||
progressive disclosure, because the agent either loads everything or loads nothing.
|
||||
`Kyberforge.PaddingPhrase` catches the common generic phrasing deterministically; other malformed
|
||||
forms are judgment.
|
||||
|
||||
**Output templates** belong in the body when the agent must emit them on every run, and in
|
||||
`references/` when only one dispatch branch produces that output. A template inlined for a branch
|
||||
most invocations never take is body-discipline padding.
|
||||
|
||||
## Auditing guidance
|
||||
|
||||
Flag as FAIL if:
|
||||
|
||||
- A Gotcha entry is a general tip or a reminder rather than a fact that defies a reasonable
|
||||
assumption
|
||||
- An inner code fence is unescaped inside a markdown block, breaking the render
|
||||
- A checklist wraps a single step
|
||||
- A conditional reference gives no trigger — `Kyberforge.PaddingPhrase` reports the common form
|
||||
- The agent must produce a specific format and no output template is given
|
||||
|
||||
Flag as SUGGESTION if:
|
||||
|
||||
- Gotchas are correctly formed but placed late in the body
|
||||
- An output template is present but permissive where the consumer needs it exact
|
||||
- A conditional reference names a trigger that is real but broader than the branch it guards
|
||||
@@ -15,7 +15,7 @@
|
||||
- **URL:** https://agentskills.io/specification.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** Complete SKILL.md format specification — frontmatter fields, constraints, body content, optional directories, progressive disclosure levels, file references, validation
|
||||
- **Contributing files:** SKILL.md, references/body-discipline.md, references/description-quality.md
|
||||
- **Contributing files:** SKILL.md, references/body-discipline.md, references/description-quality.md, references/patterns.md, references/file-structure.md, references/formatting-and-scripts.md, references/validation-scripts.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## agentskills-best-practices
|
||||
@@ -23,7 +23,7 @@
|
||||
- **URL:** https://agentskills.io/skill-creation/best-practices.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** Best practices for skill creators — starting from real expertise, spending context wisely, calibrating control, instruction patterns (gotchas, templates, checklists, validation loops)
|
||||
- **Contributing files:** SKILL.md, references/body-discipline.md
|
||||
- **Contributing files:** SKILL.md, references/body-discipline.md, references/patterns.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## agentskills-optimizing-descriptions
|
||||
@@ -47,7 +47,7 @@
|
||||
- **URL:** https://agentskills.io/skill-creation/using-scripts.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** Using scripts in skills — one-off commands, self-contained scripts with inline dependencies, designing scripts for agentic use (no interactive prompts, --help, structured output, idempotency)
|
||||
- **Contributing files:** SKILL.md
|
||||
- **Contributing files:** SKILL.md, references/formatting-and-scripts.md, references/validation-scripts.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## agentskills-quickstart
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
---
|
||||
source_keys:
|
||||
- agentskills-spec
|
||||
- agentskills-using-scripts
|
||||
---
|
||||
|
||||
# Validation Scripts Reference
|
||||
|
||||
Read this when a Step 1 script fails, cannot run, or reports something that needs interpreting.
|
||||
Nothing here is needed on a clean run.
|
||||
|
||||
## Report the gap, do not guess
|
||||
|
||||
If a script cannot run at all — Bash denied, `python3` unavailable, PyYAML not importable, `vale`
|
||||
not installed — say so as an **INFO** finding naming the script and the missing dependency, then
|
||||
fall back to the manual checks below. An INFO never changes PASS/FAIL. Silently omitting the
|
||||
dimension a script would have covered reports a clean audit that checked less than it claims to
|
||||
have checked, and the Step 4 coverage line then names a dimension nothing actually examined.
|
||||
|
||||
## Manual structural fallback
|
||||
|
||||
`validate.sh` needs `python3` **and** PyYAML, and refuses to start without either — the description
|
||||
value has to be measured after YAML folding is resolved, so skipping the ADR-0020 gates would be a
|
||||
vacuous pass rather than a partial one. The two are checked separately, so the message already names
|
||||
the right one — report it verbatim rather than diagnosing further:
|
||||
|
||||
```text
|
||||
Error: python3 is required but was not found on PATH.
|
||||
Error: PyYAML is required but is not importable by python3.
|
||||
```
|
||||
|
||||
Without them — or with Bash denied, or on a permission error — work this list
|
||||
by hand and file the results under `### Structure` exactly as the script's output would have been:
|
||||
|
||||
- **`name`** present, 1–64 characters, kebab-case (lowercase letters, digits and hyphens; no
|
||||
leading, trailing or doubled hyphen), and **matching the skill's directory name** exactly.
|
||||
- **`description`** present and non-empty; no unfilled `FILL IN:` placeholder in it. An absent or
|
||||
empty description is a **FAIL**, never a silent skip — it is the one field preloaded into every
|
||||
session, so a skill without one can never be routed to.
|
||||
- **Description length**, measured on the folded YAML value with newlines collapsed to single
|
||||
spaces — not on the raw block scalar, which counts indentation. 250 characters SUGGESTION, 400
|
||||
FAIL (ADR-0020), 1,024 FAIL (agentskills.io spec).
|
||||
- **Body length**, counting everything after the frontmatter's closing `---`. 600 words
|
||||
SUGGESTION, 900 FAIL (ADR-0020).
|
||||
- **Whole-file ceilings**, counting the file including frontmatter: 500 lines FAIL, 2,770 words
|
||||
FAIL (agentskills.io spec). These are a different measurement from the two above — report them
|
||||
as separate findings, never merged.
|
||||
- **A boundary clause is present** — either the prose form (`do not` / `instead` / `rather than` /
|
||||
`not for`) or ADR-0020's compressed `Not <thing> -> <name>` arrow. **SUGGESTION**, not FAIL:
|
||||
the absence is deterministic, but whether this skill warrants one is the auditor's call.
|
||||
- **Boundary targets resolve** — **FAIL** on a name that resolves to nothing. See the section
|
||||
below; resolving these by hand is the one item on this list with a procedure of its own.
|
||||
- **Every `references/<file>.md` named in the body exists on disk** — **FAIL**, not a suggestion.
|
||||
A dispatch table or "read X" trigger naming a missing file sends the agent nowhere. Ignore
|
||||
mentions inside fenced code blocks, and ignore a mention whose own line says the file is gone
|
||||
(`removed`, `deleted`, `renamed`, `superseded`, `replaced`, `obsolete`, `deprecated`, `former`,
|
||||
`gone`, `no longer`, `used to`) — that is a historical note, not a dispatch entry.
|
||||
- **Gotchas discipline**, both **SUGGESTION**. Locate the section by a heading that *is* Gotchas
|
||||
(`## Common Gotchas` counts; `## Gotcha handling` and `## Why gotchas matter` do not), running to
|
||||
the next heading at the same level or shallower. More than five top-level entries is one
|
||||
suggestion; a section over 25% of the body word count is a second, independent one. Count
|
||||
entries at column 0 only — an indented child bullet is not an entry — and ignore fenced code
|
||||
blocks for both.
|
||||
- **No unfilled `FILL IN:` placeholder** anywhere in the body.
|
||||
- **Every file in `scripts/`** carries the executable bit and contains no interactive prompt —
|
||||
no bare `read`, no `select`, nothing that blocks on a TTY.
|
||||
|
||||
## Resolving boundary targets by hand
|
||||
|
||||
Targets are read from **both** boundary forms. The compressed `Not <thing> -> <name>` arrow and the
|
||||
prose form are each parsed *and* target-checked, so a typo in prose phrasing fails exactly as an
|
||||
arrow typo does — do not check only the names after an arrow.
|
||||
|
||||
Build the universe by walking up **from the `SKILL.md` under audit**, never from the validator's own
|
||||
location. The nearest ancestor holding `plugins/*/.apm/skills/` or `plugins/*/.apm/agents/` is the
|
||||
authoring root, falling back to the nearest ancestor holding `.git`. When one is found the universe
|
||||
is every skill and agent under `<root>/plugins/*/`, plus the skill's own apm package, plus the
|
||||
packages that package declares in its `apm.yml` under `dependencies.apm`. Deployed `.claude/` and
|
||||
`.agents/` trees are consulted **only** when no authoring root exists — they are gitignored
|
||||
`apm install` output, and reading them would make a fresh clone and a developer machine disagree.
|
||||
|
||||
Three ways to read the result wrong:
|
||||
|
||||
- **A hyphenated name used attributively is not a dangling target.** "Use pre-commit hooks instead
|
||||
of ad-hoc scripts" reads as a route to `pre-commit` on wording alone. What separates a route from
|
||||
prose is grammar: a route target is terminal — followed by punctuation, a conjunction, or a
|
||||
boundary word — whereas a compound modifier is followed by the noun it modifies. A name followed
|
||||
by an ordinary noun still *confirms* a route when it exists, but never raises a FAIL on its own.
|
||||
- **A SUGGESTION-tier unresolved target is not a FAIL you may promote.** Terminal position alone is
|
||||
not evidence of a route: "run `pre-commit` instead", "see `commit-msg`" and "use the clean-up
|
||||
instead" are all terminal and all prose. A prose-form target earns a FAIL only when its own
|
||||
sentence names another target that *does* resolve; otherwise the script reports it and moves on,
|
||||
and so should you. Route notation — `/name` and `-> name` — is exempt and always FAILs, and it is
|
||||
the fix to recommend when the author did mean a route.
|
||||
- **`INFO boundary-target resolution DID NOT RUN` is not a pass.** The script prints it, and exits
|
||||
0, when no universe could be determined for that path — the usual cause being a skill copy
|
||||
audited outside its package. Report it as an INFO naming the unchecked targets and re-run against
|
||||
the real directory; filing it as clean signs off targets nothing verified.
|
||||
|
||||
## Script-specific failures
|
||||
|
||||
- **`validate-provenance.sh` printed nothing.** That is a pass, not a skip. It also exits 0
|
||||
silently when the skill has no `source_keys` and no `references/sources.md` — nothing to
|
||||
validate is not a finding.
|
||||
- **`vale` reports `0 files`.** Treat the pass as NOT RUN, not as clean, and fall back to full
|
||||
Step 3 judgment for the dimensions it would have covered. The bundled `Kyberforge` style is
|
||||
scoped by glob in `assets/vale/.vale.ini`; a file outside those globs is silently not linted.
|
||||
- **`E100 Runtime error ... does not exist` (exit 2) from `vale-wrap.sh`.** An explicit relative
|
||||
`--config` was passed. Pass none: the wrapper locates its own `assets/vale/.vale.ini` from its
|
||||
own path, so a resolved script path plus an unresolved config path produces exactly this. Do not
|
||||
read this exit code as vale being unavailable — that misreading sends the audit down the
|
||||
fallback path while vale was installed and working the whole time.
|
||||
- **The `vale` binary is genuinely absent** (`command not found`). Report one INFO naming it, then
|
||||
fall back to full Step 3 judgment for the description, body-discipline and patterns dimensions —
|
||||
the prefilter's whole coverage. Judge those by rubric rather than dropping them.
|
||||
- **A path argument that does not exist is a hard error** in `vale-wrap.sh`, deliberately: bare
|
||||
`vale` would fall back to reading stdin and print a clean-looking `0 errors ... in stdin`, which
|
||||
the `0 files` guard above does not catch.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -8,7 +8,14 @@ setup() {
|
||||
SCRIPT="$(cd "$BATS_TEST_DIRNAME/../scripts" && pwd)/validate.sh"
|
||||
TMPDIR="$(mktemp -d)"
|
||||
|
||||
# Helper: create a minimal valid skill directory
|
||||
# Helper: create a minimal valid skill directory.
|
||||
#
|
||||
# The description carries a boundary clause deliberately. ADR-0020's
|
||||
# missing-boundary-clause SUGGESTION fires on any description without one, so
|
||||
# a fixture that omits it is never "otherwise clean" — every test asserting
|
||||
# SUGGESTION-freedom would be asserting the boundary check's absence instead
|
||||
# of the thing it names. "anything else" is not hyphenated, so the clause adds
|
||||
# a boundary marker without adding a routing target to resolve.
|
||||
make_valid_skill() {
|
||||
local dir="$1"
|
||||
local name
|
||||
@@ -17,7 +24,7 @@ setup() {
|
||||
cat > "$dir/SKILL.md" <<EOF
|
||||
---
|
||||
name: $name
|
||||
description: A valid skill description that is well within the limit.
|
||||
description: A valid skill description that is well within the limit. Do not use for anything else.
|
||||
---
|
||||
|
||||
## Step 1
|
||||
@@ -25,6 +32,53 @@ description: A valid skill description that is well within the limit.
|
||||
Do the thing.
|
||||
EOF
|
||||
}
|
||||
|
||||
# Helper: a description of EXACTLY <n> characters that carries a boundary
|
||||
# clause and names no routing target. The tests below measure the description
|
||||
# LENGTH, so the clause has to be paid for out of the same budget rather than
|
||||
# appended to it — hence the padding arithmetic instead of a fixed suffix.
|
||||
desc_of_length() {
|
||||
python3 - "$1" <<'PY'
|
||||
import sys
|
||||
n = int(sys.argv[1])
|
||||
prefix = 'Use when doing the thing. Do not use for anything else. '
|
||||
assert n >= len(prefix), 'requested description shorter than the boundary clause'
|
||||
print(prefix + 'x' * (n - len(prefix)))
|
||||
PY
|
||||
}
|
||||
|
||||
# Helper: create a skill directory with an exact description length and an
|
||||
# exact body word count. <desc> is used verbatim; <body_words> "word"
|
||||
# tokens follow the frontmatter. Used by the ADR-0020 boundary tests.
|
||||
make_sized_skill() {
|
||||
local dir="$1" desc="$2" body_words="$3"
|
||||
local name
|
||||
name="$(basename "$dir")"
|
||||
mkdir -p "$dir"
|
||||
{
|
||||
echo "---"
|
||||
echo "name: $name"
|
||||
echo "description: $desc"
|
||||
echo "---"
|
||||
echo ""
|
||||
python3 -c "print(' '.join(['word'] * $body_words))"
|
||||
} > "$dir/SKILL.md"
|
||||
}
|
||||
|
||||
# Helper: build a self-contained fixture plugin tree so the boundary-target
|
||||
# resolver has a real authoring source to resolve against, independent of
|
||||
# this repo's live skills. Echoes the subject skill's directory.
|
||||
#
|
||||
# <root>/plugins/fixture-plugin/.apm/skills/<subject>/SKILL.md
|
||||
# <root>/plugins/fixture-plugin/.apm/skills/fixture-sibling-skill/
|
||||
# <root>/plugins/fixture-plugin/.apm/agents/fixture-sibling-agent.agent.md
|
||||
make_fixture_tree() {
|
||||
local root="$1" subject="$2"
|
||||
local apm="$root/plugins/fixture-plugin/.apm"
|
||||
mkdir -p "$apm/skills/$subject" "$apm/skills/fixture-sibling-skill" "$apm/agents"
|
||||
touch "$apm/agents/fixture-sibling-agent.agent.md"
|
||||
echo "$apm/skills/$subject"
|
||||
}
|
||||
}
|
||||
|
||||
teardown() {
|
||||
@@ -64,7 +118,7 @@ teardown() {
|
||||
assert_success
|
||||
}
|
||||
|
||||
@test "passes at exactly 1024-char description" {
|
||||
@test "the 1024-char agentskills.io spec backstop is unchanged and separate from the ADR-0020 ceiling" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
local name
|
||||
name="$(basename "$skill")"
|
||||
@@ -82,7 +136,12 @@ description: $desc
|
||||
Do the thing.
|
||||
EOF
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
# Two independent gates on one value: the spec limit still PASSES at
|
||||
# exactly 1024 (its own boundary is unmoved), while ADR-0020's 400-char
|
||||
# ceiling FAILs. The run fails on the second, not the first.
|
||||
assert_output --partial "description length 1024 chars (agentskills.io spec limit: 1024)"
|
||||
assert_output --partial "400-character ADR-0020 ceiling"
|
||||
assert_failure
|
||||
}
|
||||
|
||||
@test "passes at exactly 500 lines" {
|
||||
@@ -170,6 +229,61 @@ EOF
|
||||
assert_failure
|
||||
}
|
||||
|
||||
@test "fails when a script reads a variable with no redirect" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_valid_skill "$skill"
|
||||
printf '#!/usr/bin/env bash\nread -r ANSWER\n' > "$skill/scripts/helper.sh"
|
||||
chmod +x "$skill/scripts/helper.sh"
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
}
|
||||
|
||||
@test "fails when an interactive prompt string contains an angle bracket" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_valid_skill "$skill"
|
||||
printf '#!/usr/bin/env bash\nread -p "enter <name>: " NAME\n' > "$skill/scripts/helper.sh"
|
||||
chmod +x "$skill/scripts/helper.sh"
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
}
|
||||
|
||||
@test "passes when a script reads from a here-string" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_valid_skill "$skill"
|
||||
printf '#!/usr/bin/env bash\nLINE="a b"\nread -r X Y <<< "$LINE"\n' \
|
||||
> "$skill/scripts/helper.sh"
|
||||
chmod +x "$skill/scripts/helper.sh"
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
}
|
||||
|
||||
@test "passes when a script reads from a here-doc" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_valid_skill "$skill"
|
||||
printf '#!/usr/bin/env bash\nread -r X <<EOF\nvalue\nEOF\n' > "$skill/scripts/helper.sh"
|
||||
chmod +x "$skill/scripts/helper.sh"
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
}
|
||||
|
||||
@test "passes when a script reads from a file redirect" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_valid_skill "$skill"
|
||||
printf '#!/usr/bin/env bash\nread -r LINE < "$1"\n' > "$skill/scripts/helper.sh"
|
||||
chmod +x "$skill/scripts/helper.sh"
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
}
|
||||
|
||||
@test "passes when a script reads from a pipe continued onto the next line" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_valid_skill "$skill"
|
||||
printf '#!/usr/bin/env bash\nprintf %%s "$1" |\n read -r X\n' > "$skill/scripts/helper.sh"
|
||||
chmod +x "$skill/scripts/helper.sh"
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
}
|
||||
|
||||
@test "fails when name contains consecutive hyphens" {
|
||||
local skill="$TMPDIR/my--skill"
|
||||
make_valid_skill "$skill"
|
||||
@@ -196,3 +310,261 @@ EOF
|
||||
run bash "$SCRIPT"
|
||||
assert_failure
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ADR-0020 — description budget (250 SUGGESTION / 400 FAIL)
|
||||
#
|
||||
# These sit UNDER the agentskills.io 1024-character spec backstop above, which
|
||||
# is unchanged. Both ceilings are inclusive: exactly at the number passes that
|
||||
# tier, one past it trips.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@test "ADR-0020: description of exactly 250 chars raises no suggestion" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_sized_skill "$skill" "$(desc_of_length 250)" 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
refute_output --partial "SUGGESTION"
|
||||
}
|
||||
|
||||
@test "ADR-0020: description of 251 chars raises a SUGGESTION and still exits 0" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_sized_skill "$skill" "$(desc_of_length 251)" 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
assert_output --partial "SUGGESTION"
|
||||
assert_output --partial "description is 251 chars"
|
||||
assert_output --partial "All checks passed (1 suggestion(s))."
|
||||
}
|
||||
|
||||
@test "ADR-0020: description of exactly 400 chars is a SUGGESTION, not a FAIL" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_sized_skill "$skill" "$(desc_of_length 400)" 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
assert_output --partial "SUGGESTION"
|
||||
}
|
||||
|
||||
@test "ADR-0020: description of 401 chars FAILs and exits non-zero" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_sized_skill "$skill" "$(desc_of_length 401)" 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
assert_output --partial "description is 401 chars"
|
||||
assert_output --partial "400-character ADR-0020 ceiling"
|
||||
}
|
||||
|
||||
@test "ADR-0020: description length is measured after YAML folding is resolved" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
mkdir -p "$skill"
|
||||
# A >-folded block scalar: 11 lines of 40 chars folded with 10 joining
|
||||
# spaces = 450 characters. Measured off its raw `description: >` line it is
|
||||
# 1 character and passes; measured as the folded VALUE it must FAIL. This
|
||||
# is exactly the case a line-wise regex gets wrong.
|
||||
{
|
||||
echo "---"
|
||||
echo "name: my-skill"
|
||||
echo "description: >"
|
||||
python3 -c "print('\n'.join([' ' + 'x' * 40] * 11))"
|
||||
echo "---"
|
||||
echo ""
|
||||
echo "Do the thing."
|
||||
} > "$skill/SKILL.md"
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
assert_output --partial "description is 450 chars"
|
||||
assert_output --partial "400-character ADR-0020 ceiling"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ADR-0020 — body budget (600 SUGGESTION / 900 FAIL), body ONLY
|
||||
#
|
||||
# Distinct from the 2,770-word whole-file spec ceiling above, which counts
|
||||
# frontmatter too and is unchanged. Do not unify them.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@test "ADR-0020: body of exactly 600 words raises no suggestion" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_sized_skill "$skill" "A short valid description. Do not use for anything else." 600
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
refute_output --partial "SUGGESTION"
|
||||
}
|
||||
|
||||
@test "ADR-0020: body of 601 words raises a SUGGESTION and still exits 0" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_sized_skill "$skill" "A short valid description. Do not use for anything else." 601
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
assert_output --partial "body is 601 words"
|
||||
assert_output --partial "All checks passed (1 suggestion(s))."
|
||||
}
|
||||
|
||||
@test "ADR-0020: body of exactly 900 words is a SUGGESTION, not a FAIL" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_sized_skill "$skill" "A short valid description. Do not use for anything else." 900
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
assert_output --partial "body is 900 words"
|
||||
}
|
||||
|
||||
@test "ADR-0020: body of 901 words FAILs and exits non-zero" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
make_sized_skill "$skill" "A short valid description. Do not use for anything else." 901
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
assert_output --partial "body is 901 words"
|
||||
assert_output --partial "900-word ADR-0020 ceiling"
|
||||
}
|
||||
|
||||
@test "ADR-0020: the body gate counts the body only — frontmatter words do not count toward it" {
|
||||
local skill="$TMPDIR/my-skill"
|
||||
# 895 body words plus a description long enough that the WHOLE FILE is well
|
||||
# over 900 words. The body gate must stay silent; the 2,770-word whole-file
|
||||
# ceiling is a separate measurement and is nowhere near tripping.
|
||||
make_sized_skill "$skill" "$(python3 -c "print(' '.join(['w'] * 100))")" 895
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
refute_output --partial "900-word ADR-0020 ceiling"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# ADR-0020 — resolvable boundary targets
|
||||
#
|
||||
# Resolved against the AUTHORING SOURCE (plugins/*/.apm/skills/ and
|
||||
# plugins/*/.apm/agents/), never .claude/skills/, so the check works offline and
|
||||
# before an apm install. Every fixture below builds its own plugin tree rather
|
||||
# than leaning on this repo's live skills.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@test "ADR-0020: a boundary target naming an existing sibling skill resolves" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use fixture-sibling-skill instead." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
assert_output --partial "boundary target(s) resolve"
|
||||
}
|
||||
|
||||
@test "ADR-0020: a boundary target naming a non-existent skill FAILs when its sentence names one that resolves" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
# `fixture-sibling-skill` is the corroborator: a prose-form target only earns
|
||||
# a FAIL when its own sentence proves it is a routing sentence. See the
|
||||
# shared resolver's CORROBORATION note, and the uncorroborated case below.
|
||||
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use fixture-sibling-skill or fixture-missing-skill instead." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
assert_output --partial "routes to 'fixture-missing-skill'"
|
||||
}
|
||||
|
||||
@test "ADR-0020: a LONE boundary target naming a non-existent skill is a SUGGESTION, not a FAIL" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
# Same grammar as the case above and as "run \`pre-commit\` instead" — a
|
||||
# route verb, a hyphenated name, terminal position. Nothing local separates a
|
||||
# broken route from a tool name, so the target is named on every run but does
|
||||
# not block: this gate ships with no baseline and no suppression mechanism.
|
||||
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use fixture-missing-skill instead." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
assert_output --partial "SUGGESTION"
|
||||
assert_output --partial "routes to 'fixture-missing-skill'"
|
||||
refute_output --partial "FAIL description routes to"
|
||||
}
|
||||
|
||||
@test "ADR-0020: a boundary target naming an AGENT file resolves (agents are valid routing targets)" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
make_sized_skill "$skill" "Use when doing the thing. Do not use when the caller is an agent — invoke fixture-sibling-agent instead." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
assert_output --partial "boundary target(s) resolve"
|
||||
}
|
||||
|
||||
@test "ADR-0020: a /slash-command boundary target that does not resolve FAILs" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
make_sized_skill "$skill" "Use when doing the thing. Do not use when improvements are wanted — use /fixture-missing-improve instead." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
assert_output --partial "routes to 'fixture-missing-improve'"
|
||||
}
|
||||
|
||||
@test "ADR-0020: a backticked name that does not resolve FAILs when its sentence names one that resolves" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
make_sized_skill "$skill" "Use when doing the thing. Composes \`fixture-sibling-skill\` and \`fixture-missing-helper\` for the shared part." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
assert_output --partial "routes to 'fixture-missing-helper'"
|
||||
}
|
||||
|
||||
@test "ADR-0020: a /slash-command target is route NOTATION and FAILs on its own, uncorroborated" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
# The escape hatch from the SUGGESTION tier: `/name` and `-> name` are never
|
||||
# how prose cites a tool, so they are exempt from corroboration. An author
|
||||
# who wants a route checked unconditionally writes one of those two forms.
|
||||
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use /fixture-missing-notation instead." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
assert_output --partial "routes to 'fixture-missing-notation'"
|
||||
}
|
||||
|
||||
@test "ADR-0020: a bare hyphenated word outside a boundary sentence is not read as a routing target" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
# "run pre-commit hooks" is pc-run's real phrasing. A naive extractor reads
|
||||
# it as a route to a non-existent `pre-commit` skill.
|
||||
make_sized_skill "$skill" "Use when the user wants to run pre-commit hooks or install git hooks." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
refute_output --partial "pre-commit"
|
||||
}
|
||||
|
||||
@test "ADR-0020: an arrow chain outside a boundary clause is not read as a routing target" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
# diagnose's real process chain. Only ADR-0020's `Not <thing> -> <skill>`
|
||||
# form makes a bare arrow target a route.
|
||||
make_sized_skill "$skill" "Reproduce → minimise → instrument → fix → regression-test. Use when a bug is reported." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
refute_output --partial "regression-test"
|
||||
}
|
||||
|
||||
@test "ADR-0020: MCP tool names and capitalised tool names are not read as routing targets" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
make_sized_skill "$skill" "Use when writing issues. Do not use for local files (use Read/Write/Edit) — that write goes through \`issue_write\`/\`pull_request_write\` instead." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
refute_output --partial "routes to"
|
||||
}
|
||||
|
||||
@test "ADR-0020: ADR's compressed boundary form (Not <thing> -> <skill>) is checked" {
|
||||
local skill
|
||||
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
|
||||
make_sized_skill "$skill" "Use when doing the thing. Not the other thing → fixture-missing-target." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_failure
|
||||
assert_output --partial "routes to 'fixture-missing-target'"
|
||||
}
|
||||
|
||||
@test "ADR-0020: the boundary check declines rather than false-FAILs when no authoring source is found" {
|
||||
# Deliberately NOT built with make_fixture_tree: this skill sits in a bare
|
||||
# temp directory with no plugins/*/.apm/ above it and no .git, so the resolver
|
||||
# legitimately has no universe. That is a real path (a skill being drafted
|
||||
# outside any repo), and the required behaviour is to DECLINE OUT LOUD rather
|
||||
# than either false-FAIL or pass in silence — silence is what let a whole gate
|
||||
# family go missing unnoticed. So the INFO text and the named unchecked target
|
||||
# are both asserted, not just the absence of a failure.
|
||||
local skill="$TMPDIR/orphan/my-skill"
|
||||
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use some-other-skill instead." 10
|
||||
run bash "$SCRIPT" "$skill"
|
||||
assert_success
|
||||
refute_output --partial "routes to"
|
||||
assert_output --partial "boundary-target resolution DID NOT RUN"
|
||||
assert_output --partial "Unchecked target(s): some-other-skill"
|
||||
}
|
||||
|
||||
@@ -6,6 +6,14 @@ Author and refine skills conforming to the [agentskills.io](https://agentskills.
|
||||
|
||||
Routes to one of two flows based on context: if no skill directory exists at the target path, it scaffolds the directory from annotated templates, fills in `SKILL.md` and supporting files, and validates the result. If an existing skill directory and improvement signals are both present, it groups those signals by root cause and applies targeted edits, then re-validates. In both flows, bumps the skill's `metadata.version` when present (minor for create, patch for improve).
|
||||
|
||||
`SKILL.md` itself carries only the dispatch table, the invocation-axis decision, the contract gates and the shared close; each flow lives in its own self-contained reference file, per ADR-0020.
|
||||
|
||||
## The contract it teaches
|
||||
|
||||
Authored skills are held to the ADR-0020 context budget. A description carries a trigger clause, at most one capability clause, and a boundary clause of the form `Not <thing> -> <skill-name>` whose target must resolve to a real skill or agent — 250 characters target, 400 hard ceiling. A body carries the decision procedure only — 600 words target, 900 hard ceiling, counting the body alone, which is a separate measurement from the 2,770-word / 500-line whole-file spec backstop. Skills with two or more mutually exclusive flows must dispatch. `references/contract.md` holds the full rules; `assets/templates/SKILL.md` encodes them as a fill-in skeleton.
|
||||
|
||||
Before a description is written, the skill asks whether the target is model-invoked or hand-invoked. A hand-invoked skill sets `disable-model-invocation: true` and carries one plain human-facing sentence with no trigger list.
|
||||
|
||||
## Before you start
|
||||
|
||||
- Run `/grill-me` to resolve design decisions before creating a new skill
|
||||
@@ -14,7 +22,7 @@ Routes to one of two flows based on context: if no skill directory exists at the
|
||||
|
||||
## Placement
|
||||
|
||||
`scripts/new-skill.sh` resolves the mode automatically by walking up from the given path — see `SKILL.md` Step 1 for the full algorithm.
|
||||
`scripts/new-skill.sh` resolves the mode automatically by walking up from the given path — see `references/create.md` Step 1 for the full algorithm.
|
||||
|
||||
| Mode | Path | Chosen when |
|
||||
|------|------|-------------|
|
||||
@@ -36,18 +44,30 @@ If the destination resolves inside an APM package, read `references/deployment-m
|
||||
| `README.md` | Human-readable overview of the skill and its files |
|
||||
| `SKILL.md` | Skill instructions for agents |
|
||||
| `scripts/new-skill.sh` | Walks up from the given path to resolve package vs standalone mode, then copies annotated templates to the resolved destination |
|
||||
| `references/create.md` | The create flow end to end — prerequisites, package-intent gate, scaffold, frontmatter, scripts, references, sources (loaded on demand) |
|
||||
| `references/improve.md` | The improve flow end to end — signal verification, root-cause grouping, announcement, edits (loaded on demand) |
|
||||
| `references/contract.md` | The ADR-0020 description and body contract, the Gotchas constraint, the two size gates, body patterns, and org-policy embedding (loaded on demand) |
|
||||
| `references/retrofit.md` | Bringing a pre-ADR-0020 skill into contract — ordered cut procedure, the mutually-exclusive-flows test, reference-file conventions, the collateral checklist, and a worked description retrofit (loaded from the improve flow when a budget is exceeded) |
|
||||
| `references/deployment-modes.md` | APM package vs standalone differences and self-containment/cache-isolation rules (loaded on demand) |
|
||||
| `references/scripts.md` | Package runners, inline dependency patterns, and full script contract (loaded on demand) |
|
||||
| `references/sources.md` | Upstream research sources and which skill files each contributed to |
|
||||
| `assets/templates/SKILL.md` | Annotated SKILL.md template |
|
||||
| `assets/templates/SKILL.md` | Annotated SKILL.md template — emits an ADR-0020-compliant description and body skeleton |
|
||||
| `assets/templates/README.md` | Annotated README template for the new skill |
|
||||
| `assets/templates/scripts/README.md` | Placeholder for bundled scripts |
|
||||
| `assets/templates/references/README.md` | Placeholder for reference docs |
|
||||
| `assets/templates/references/sources.md` | Sources provenance template for new skills |
|
||||
| `assets/templates/assets/README.md` | Placeholder for static assets |
|
||||
| `assets/templates/tests/README.md` | Placeholder for test files |
|
||||
| `tests/new-skill.bats` | Bats test suite for `scripts/new-skill.sh` |
|
||||
| `tests/README.md` | Setup instructions for bats-support and bats-assert test dependencies |
|
||||
| `tests/new-skill.bats` | (source-only) Bats test suite for `scripts/new-skill.sh` |
|
||||
| `tests/README.md` | (source-only) Setup instructions for bats-support and bats-assert test dependencies |
|
||||
|
||||
Rows marked **(source-only)** exist in the authoring source (`.apm/skills/skill-author/`) but are
|
||||
not present in an installed plugin: `scripts/sync-plugin-content.sh` strips
|
||||
`<category>/<name>/tests` when it generates the flat mirror, because these are dev-time fixtures no
|
||||
plugin host needs to discover (ADR-0017). Run them from a repo checkout, not from an install. The
|
||||
`assets/templates/tests/README.md` row above is **not** source-only — the exclusion is depth-scoped
|
||||
to `<category>/<name>/tests`, so the scaffolding template tree ships intact, which
|
||||
`scripts/new-skill.sh` depends on at runtime.
|
||||
|
||||
## Spec reference
|
||||
|
||||
|
||||
@@ -1,14 +1,9 @@
|
||||
---
|
||||
name: skill-author
|
||||
description: >
|
||||
Use when the user wants to create a new skill from scratch ("write a skill
|
||||
for X", "build a skill that does Y", "create a SKILL.md for Z") or improve
|
||||
an existing one ("improve this skill", "fix based on feedback", "apply these
|
||||
audit findings", "update based on grill output"). Also use when the user provides inline feedback
|
||||
about a skill's behavior and wants it applied, or when a grill session, eval
|
||||
run, or audit has produced findings the user wants acted on — even if they
|
||||
don't say "improve" explicitly. Do not use for read-only review — use
|
||||
/skill-audit instead. Do not use to author agent definition files.
|
||||
Use when the user wants to create a new skill from scratch, or apply audit
|
||||
findings, grill output, eval results, or inline feedback to an existing one.
|
||||
Not read-only review -> `skill-audit`. Not agent files -> `agent-author`.
|
||||
allowed-tools: Bash Read Write Edit
|
||||
metadata:
|
||||
category: factory
|
||||
@@ -24,283 +19,43 @@ metadata:
|
||||
|
||||
## Gotchas
|
||||
|
||||
- Patching per symptom is the default failure mode. Three eval failures may all trace to one missing instruction — always identify the root cause before editing.
|
||||
- Do not create new scripts unless a signal explicitly calls for it. Writing scripts from scratch requires transcript analysis that is out of scope here; flag the opportunity as a suggestion instead.
|
||||
- Never spawn a subagent to audit or recheck your own work during an authoring pass. Run `/skill-audit` yourself, inline, in the same context as the edits you just made. A *separate* independent recheck via a clean-context subagent is the `/forge` skill's outer-loop responsibility exclusively — delegating it inward here duplicates that layer and introduces a race: a stray self-spawned subagent can have its worktree torn down by concurrent cleanup, destroying an uncommitted draft before it was ever safe.
|
||||
- The word gates are two measurements, not two tiers of one rule: the 2,770-word / 500-line spec backstop counts the whole file, Step 3's gate the body alone. Never unify them.
|
||||
- Never spawn a subagent to audit or recheck your own work — run `/skill-audit` inline, in the same context as the edits. Clean-context recheck belongs to `/forge`'s outer loop, and a self-spawned subagent's worktree can be torn down by concurrent cleanup, destroying an uncommitted draft.
|
||||
- Do not create new scripts unless a signal explicitly calls for it. Writing one from scratch requires out-of-scope transcript analysis — flag the opportunity as a suggestion instead.
|
||||
|
||||
## Route
|
||||
## Step 1 — Dispatch
|
||||
|
||||
Determine which flow to follow before touching the filesystem:
|
||||
| Condition | Flow | Reference |
|
||||
|---|---|---|
|
||||
| No skill directory at the target path | Create | `references/create.md` |
|
||||
| Directory exists, at least one improvement signal present | Improve | `references/improve.md` |
|
||||
| Directory exists, no signals | Stop and ask | — |
|
||||
|
||||
- **No skill directory at the target path** → follow **Creating a new skill**
|
||||
- **Directory exists + at least one improvement signal present** → follow **Improving an existing skill**
|
||||
- **Directory exists + no signals present** → ask: "No improvement signals found. Did you mean to create a new skill, or do you have feedback to apply?"
|
||||
Signals: grill output, `/skill-audit` findings, inline feedback, eval results, session context describing what went wrong. With none, ask whether the user meant to create a new skill or has feedback to apply.
|
||||
|
||||
Signals include: grill session output, `/skill-audit` findings (PASS/FAIL punch list), inline user feedback, session context describing what went wrong.
|
||||
Read only the reference matching the resolved flow — each is self-contained. If the target sits inside a git worktree, capture `git log --oneline -1` before touching the filesystem; Step 4 needs it.
|
||||
|
||||
**Before running the scaffold script**, judge whether the destination is meant to be inside an APM package — the script can't tell "no package here" apart from "package not scaffolded yet":
|
||||
## Step 2 — Invocation axis
|
||||
|
||||
- Package intent but no `type:`-bearing `apm.yml` found at/above the destination (e.g. "add to my apm package", or a sibling `.apm/`/`apm.yml` exists nearby) → **stop**, tell the user to run `/apm-workflow configure` (`apm plugin init`, from inside the package directory) first, then retry. Don't fall through to standalone mode.
|
||||
- Otherwise (a `~/`-rooted destination, or no package context implied) → run `scripts/new-skill.sh`; it resolves package vs. standalone automatically (see Step 1).
|
||||
Decide before writing any description: model-invoked or hand-invoked?
|
||||
|
||||
## Creating a new skill
|
||||
- **Hand-invoked** — the user types `/name` and no agent should route to it. Set `disable-model-invocation: true` and write one plain human-facing sentence: no trigger list, no boundary clause. Skip Step 3's description rules.
|
||||
- **Model-invoked** — the default.
|
||||
|
||||
### Prerequisites
|
||||
## Step 3 — Contract
|
||||
|
||||
Run `/grill-me` on the skill's design and research the target domain first.
|
||||
Share those outputs in this conversation: grill context, research docs, examples, constraints.
|
||||
Before writing or editing a description, or restructuring a body, read `references/contract.md` — the banned-content list, boundary form, include/exclude rubric and body patterns.
|
||||
|
||||
Design for one coherent user intent — skills too narrow force multiple loads per task; too broad are hard to activate precisely.
|
||||
Gates `/skill-audit` enforces in both flows:
|
||||
|
||||
**Before touching the filesystem, verify you have:**
|
||||
- [ ] A clear purpose — what specific task will this skill handle?
|
||||
- [ ] Trigger scenarios — when should an agent activate it, including indirect cases?
|
||||
- [ ] Skill name (kebab-case) and destination path
|
||||
- [ ] Capture `git log --oneline -1` now, before touching the filesystem — Step 7 needs it to verify a real commit landed
|
||||
- **Description** — a trigger clause, at most one capability clause, and a boundary clause shaped `Not <thing> -> <skill-name>` whose target resolves to a real skill or agent. 250 characters SUGGESTION, 400 FAIL, value only.
|
||||
- **Body** — decision procedure only: ordered steps, branches, gates, and which reference to load when. 600 words SUGGESTION, 900 FAIL, body only. At two or more mutually exclusive flows a dispatch table is mandatory and each flow gets its own self-contained `references/` file.
|
||||
- **Gotchas** — each contradicting a reasonable default. A Gotcha paraphrasing a step below it is a FAIL; over five entries is a SUGGESTION only.
|
||||
|
||||
If any are missing, stop and ask the user before proceeding.
|
||||
## Step 4 — Validate and close
|
||||
|
||||
**Requires `/skill-audit`** — used in Step 7 for final validation. Both skills ship in the kyberforge plugin and are co-installed. If `/skill-audit` is unavailable, stop and ask the user to install the kyberforge plugin before continuing.
|
||||
Run `/skill-audit` on the resolved skill directory; resolve every FAIL before reporting done. It checks name-to-directory match, placeholders, both size budgets, boundary-target resolution and script hygiene — do not hand-check those. Hand-check the one thing it misses: an empty body reports `PASS SKILL.md body word count 0 (ADR-0020 target: 600)`, so confirm at least one non-empty section exists.
|
||||
|
||||
### Step 1 — Scaffold
|
||||
With `metadata.version` present, bump the **minor** version on create (new skills start at `0.1.0`) and the **patch** version on improve.
|
||||
|
||||
Run the copy script with the skill name and a path inside or at the target:
|
||||
|
||||
```bash
|
||||
bash scripts/new-skill.sh <skill-name> <path>
|
||||
```
|
||||
|
||||
The script walks up from `<path>` for a package boundary: an ancestor `apm.yml` with a top-level `type:` field (`instructions`/`skill`/`hybrid`/`prompts`) means **package mode** — scaffolds into `<package-root>/.apm/skills/<skill-name>/`, not under `<path>` (a subdirectory of the package works fine as `<path>`). A `type:`-less `apm.yml` is a marketplace-only manifest, skipped. Hitting `.git` or the filesystem root first means **standalone mode** — scaffolds directly into `<path>/<skill-name>/`, same as before.
|
||||
|
||||
Examples:
|
||||
```bash
|
||||
# Package mode — packages/my-pkg/apm.yml already has `type: skill`
|
||||
bash scripts/new-skill.sh my-tool packages/my-pkg/
|
||||
|
||||
# Standalone mode — no apm.yml/.git above ~/.agents/skills/
|
||||
bash scripts/new-skill.sh my-tool ~/.agents/skills/
|
||||
```
|
||||
|
||||
The script prints which mode it used and where the skill landed — read its output.
|
||||
|
||||
In package mode, read `references/deployment-modes.md` before adding any file references to SKILL.md.
|
||||
|
||||
### Step 2 — Update `apm.yml` includes (package mode only)
|
||||
|
||||
Skip in standalone mode. In package mode, check the resolved package's `apm.yml`: if `includes:` is an explicit list (not `auto`), append `.apm/skills/<skill-name>/` to it if not already present, preserving YAML formatting. If `includes: auto` or the field is absent, do nothing — `auto` already covers the new skill. Use Read/Edit directly on `apm.yml`; this isn't part of `scripts/new-skill.sh`.
|
||||
|
||||
### Step 3 — Fill in SKILL.md
|
||||
|
||||
Open the new skill's `SKILL.md` (the path Step 1 printed). Replace every `FILL IN:` placeholder.
|
||||
|
||||
**Frontmatter**
|
||||
|
||||
**`name`** — already set by the scaffold script. Must exactly match the directory name. Format: 1–64 characters, lowercase letters/numbers/hyphens only, no leading, trailing, or consecutive hyphens (`--`).
|
||||
|
||||
**`description`** — carries the entire triggering burden. Rules:
|
||||
- Imperative: "Use when..." not "This skill..."
|
||||
- Focus on user intent, not implementation — describe what the user is trying to achieve, not the skill's internal mechanics
|
||||
- Specific about capabilities ("parses and validates OpenAPI specs", not "helps with APIs")
|
||||
- Include indirect triggers: "even if the user doesn't mention X explicitly"
|
||||
- Add "Do not use when..." only if a near-miss skill exists that could steal activations
|
||||
- Hard limit: 1024 characters — count before finalizing
|
||||
|
||||
**Optional fields** — uncomment and fill in or remove entirely:
|
||||
- `license` — include when distributing the skill externally
|
||||
- `compatibility` — include if the skill requires specific tools, runtimes, or network access (max 500 characters)
|
||||
- `metadata` — key-value map; use `author`, `version`, `category`; add `source_keys` now (see below) if research sources are in context
|
||||
- `allowed-tools` — space-separated pre-approved tools; reduces permission prompts (experimental — support varies by client)
|
||||
|
||||
**`metadata.source_keys`** — if research sources are in context, list the relevant slugs here as you write the body; don't defer this to Step 6. Agents that fill in source_keys late tend to omit it entirely. Example:
|
||||
```yaml
|
||||
metadata:
|
||||
source_keys:
|
||||
- my-source-slug
|
||||
- another-slug
|
||||
```
|
||||
|
||||
**Embedding org-specific policy** — if a skill encodes a rule sourced from an org convention file (e.g. `core/instructions/*.md`), inline that content directly into the skill (SKILL.md or a `references/` file) rather than pointing to the file's path. Plugins must be self-contained and portable — the org file may not exist wherever the plugin is installed, and in this repo such files are meant to be deleted once their content is fully embedded downstream. Tag the inlined content with a `source_keys` entry using the same `references/sources.md` schema as Step 6, noting in the `Research doc:` field that the source is an org convention rather than a plugin research corpus entry, so provenance survives after the source file is gone.
|
||||
|
||||
**Body — include only what the agent lacks**
|
||||
|
||||
Rename the placeholder section heading to one that fits the skill's structure — `## Step 1`, `## Workflow`, `## Instructions`, etc.
|
||||
|
||||
Ask of every sentence: "Would the agent get this wrong without it?" Cut anything that answers "no."
|
||||
|
||||
**Include:**
|
||||
- Non-obvious sequences or ordering constraints — the agent may skip or reorder steps without this
|
||||
- Domain conventions the agent cannot infer from general knowledge — this is the core value a skill adds
|
||||
- One default per decision point, plus one escape hatch — never a menu; menus cause the agent to pause or pick arbitrarily
|
||||
- Gotchas — facts that defy reasonable assumptions; the agent will get these wrong every time without them
|
||||
|
||||
**Exclude:**
|
||||
- Concepts the agent already knows (what JSON is, how HTTP works) — adds tokens without changing behavior
|
||||
- Exhaustive option lists — pick a default; the agent doesn't benefit from choosing
|
||||
- Steps the agent handles independently — over-specifying leads agents to follow unproductive paths
|
||||
- Restatements of the description — it's already in context; repeating it wastes the token budget
|
||||
|
||||
**Patterns**
|
||||
|
||||
**Gotchas** — highest value; place near the top:
|
||||
````markdown
|
||||
## Gotchas
|
||||
- <Fact that defies a reasonable assumption>
|
||||
- <Non-obvious naming discrepancy or hidden constraint>
|
||||
````
|
||||
|
||||
**Default with escape hatch** (not a menu):
|
||||
````markdown
|
||||
Use <X> for <task>. For <edge case>, use <Y> instead.
|
||||
````
|
||||
|
||||
**Prescriptive sequence** (when order is critical or fragile):
|
||||
````markdown
|
||||
Run exactly:
|
||||
```bash
|
||||
<command>
|
||||
```
|
||||
Do not modify flags.
|
||||
````
|
||||
|
||||
**Checklist** (multi-step workflows):
|
||||
````markdown
|
||||
- [ ] Step 1: ...
|
||||
- [ ] Step 2: ...
|
||||
````
|
||||
|
||||
**Conditional reference** (progressive disclosure — load only when needed):
|
||||
````markdown
|
||||
If <condition>, read `references/<file>.md`.
|
||||
````
|
||||
|
||||
**Output format template** (when the skill produces structured output):
|
||||
````markdown
|
||||
Output format:
|
||||
```
|
||||
<field>: <value>
|
||||
<field>: <value>
|
||||
```
|
||||
````
|
||||
For longer templates, place in `assets/<name>.md` and reference conditionally.
|
||||
|
||||
**Size budget**
|
||||
|
||||
Keep `SKILL.md` under 500 lines; 5,000 tokens is the recommended body budget. When approaching the limit:
|
||||
- Move reference material to `references/<topic>.md` and load it conditionally
|
||||
- Bundle repeated executable logic into `scripts/` rather than reinventing each run
|
||||
|
||||
### Step 4 — Add scripts (if needed)
|
||||
|
||||
Place executable scripts in `scripts/`. Critical rule: **no interactive prompts** — agents run non-interactive; blocking on TTY input hangs indefinitely. Accept all input via flags, env vars, or stdin.
|
||||
|
||||
If adding a script, read `references/scripts.md` first — it covers the full contract: structured output, pinned versions, self-contained deps, idempotency, exit codes, dry-run, error messages, and output size limits.
|
||||
|
||||
If no scripts are needed, delete `scripts/README.md` and the `scripts/` directory.
|
||||
|
||||
### Step 5 — Add references, assets, and tests (if needed)
|
||||
|
||||
**`references/`** — additional documentation loaded on demand. One topic per file.
|
||||
Reference conditionally from SKILL.md: `If <condition>, read references/<file>.md`.
|
||||
Keep reference chains one level deep — a reference file that references another reference file is rarely loaded correctly.
|
||||
|
||||
**`assets/`** — static resources: templates, schemas, lookup tables.
|
||||
Reference by relative path from SKILL.md.
|
||||
|
||||
**`tests/`** — test files for scripts in `scripts/`. Use when scripts are complex
|
||||
enough to break silently. Test infrastructure (`.bats`, `*_test.*`) belongs here,
|
||||
not in `scripts/`. See `tests/README.md` for setup instructions.
|
||||
|
||||
If not needed, delete the placeholder READMEs and their directories.
|
||||
|
||||
### Step 6 — Populate or delete `references/sources.md`
|
||||
|
||||
If a research `sources.md` is present in the conversation context:
|
||||
|
||||
1. Read it and filter to entries with `` `extracted` `` status only.
|
||||
2. For each entry, determine which skill files it contributed to (SKILL.md and any files in references/ that drew from it). Update `Contributing files` accordingly — list skill files, not research topic files.
|
||||
3. Write the updated content to `references/sources.md`. For each entry, include `- **Research doc:** <path>` where `<path>` is the relative path from the repo root to the plugin-level research sources file this entry was drawn from (e.g. `plugins/myplugin/docs/research/docs/<topic>/sources.md`). This field is required on every entry — it makes the provenance chain explicit and is validated by `/skill-audit`.
|
||||
4. Add `source_keys` to the frontmatter of `SKILL.md` (under `metadata`) listing the slugs of sources that informed it.
|
||||
5. For each file in `references/` that was informed by research sources, add `source_keys` frontmatter (same format as research topic files) listing the relevant slugs.
|
||||
|
||||
If no research `sources.md` is in context, delete `references/sources.md`.
|
||||
|
||||
### Step 7 — Validate and close
|
||||
|
||||
Before running the audit, confirm:
|
||||
- [ ] Skill name matches the directory name exactly
|
||||
- [ ] `description` field is present and non-empty
|
||||
- [ ] Body has at least one non-empty section
|
||||
- [ ] No `FILL IN:` placeholders remain in any file
|
||||
|
||||
Run `/skill-audit` on the skill directory Step 1 reported — either `<package-root>/.apm/skills/<skill-name>/` or `<path>/<skill-name>/`.
|
||||
|
||||
All FAIL findings must be resolved before the skill is considered done.
|
||||
|
||||
If the skill is versioned (`metadata.version`), set it to the next **minor** version (e.g. `0.2.0` → `0.3.0`). New skills without a prior version start at `0.1.0`.
|
||||
|
||||
**Commit verification.** Capture `git log --oneline -1` before Step 1 and keep it. Once the audit is clean, run `git add` and `git commit` for the new skill files — do not stop at staging. Then run `git log --oneline -1` again and confirm the hash changed from the one you captured at the start. A non-empty `git diff --stat` is not sufficient proof of completion: staged-but-uncommitted work isn't part of any commit and can be silently lost if the working tree is cleaned up before a commit lands. Only report the skill as done once the hash has actually changed.
|
||||
|
||||
## Improving an existing skill
|
||||
|
||||
### Step 1 — Verify inputs
|
||||
|
||||
Confirm the skill directory path exists and that at least one improvement signal is present in the conversation or a referenced file.
|
||||
|
||||
If the skill dir is missing, ask for it. If no signals are present, stop: "This skill applies existing signals to a skill. For a blind review without signals, use `/skill-audit` instead."
|
||||
|
||||
Capture `git log --oneline -1` now, before making any edits — Step 5 needs it to verify a real commit landed.
|
||||
|
||||
Signals can come from anywhere in the conversation or referenced files:
|
||||
- Grill session output (most common predecessor in the factory sequence)
|
||||
- `/skill-audit` findings (PASS/FAIL/SUGGESTION punch list)
|
||||
- Human feedback (feedback.json, inline in conversation, PR or issue comments)
|
||||
- Session context describing what went wrong
|
||||
|
||||
Also verify the `name` field in frontmatter matches the skill's directory name exactly.
|
||||
|
||||
### Step 2 — Gather and group signals
|
||||
|
||||
Read the current skill files (SKILL.md and any files in scripts/, references/, assets/, tests/). Then collect all signals from the conversation and any file paths the user has referenced.
|
||||
|
||||
Group signals by **root cause**, not symptom. Ask: "What single gap in the skill causes this cluster of failures?" One root cause → one fix. Do not make a separate edit for each symptom.
|
||||
|
||||
```text
|
||||
Example:
|
||||
- Session context: output format is wrong on every run
|
||||
- Audit finding: no output template defined
|
||||
- User feedback: "I always have to ask it to format the output"
|
||||
→ Root cause: SKILL.md has no output format specification → one fix: add an output template
|
||||
```
|
||||
|
||||
### Step 3 — Announce planned changes
|
||||
|
||||
Before editing, state:
|
||||
- Which root causes were identified and what evidence supports each
|
||||
- Which files will be changed and what will change in each
|
||||
|
||||
Then proceed — edits are reversible via git, no approval checkpoint needed.
|
||||
|
||||
### Step 4 — Apply changes
|
||||
|
||||
Edit any file in the skill directory that the signals point to: SKILL.md, scripts/, references/, assets/, tests/, README.md.
|
||||
|
||||
**Generalize, don't patch.** Find the underlying gap, not the specific example that failed. A fix scoped only to the test cases you've seen will overfit and perform worse on new inputs.
|
||||
|
||||
**Keep it lean.** Remove instructions that aren't pulling their weight. For every sentence you add, ask: "Would the agent get this wrong without it?" A shorter, focused skill consistently outperforms an exhaustive one.
|
||||
|
||||
**Explain the why.** Reasoning-based instructions outperform rigid directives. If you find yourself writing a rule in all caps (ALWAYS/NEVER), reframe it: explain why the behavior matters so the agent can apply judgment in edge cases.
|
||||
|
||||
If a signal points to a script or reference file, edit that file directly rather than adding a workaround in SKILL.md.
|
||||
|
||||
### Step 5 — Validate and close
|
||||
|
||||
Before running the audit, confirm:
|
||||
- [ ] Skill name still matches the directory name
|
||||
- [ ] No `FILL IN:` placeholders were introduced
|
||||
- [ ] No previously-passing audit checks were broken by the edits
|
||||
|
||||
Run `/skill-audit` on the skill directory. Resolve any FAIL findings before considering the improvement complete.
|
||||
|
||||
If the skill is versioned (`metadata.version`), bump the **patch** version (e.g. `0.1.0` → `0.1.1`).
|
||||
|
||||
**Commit verification.** Capture `git log --oneline -1` at the start of Step 1 and keep it. Once the audit is clean, run `git add` and `git commit` for the changed files — do not stop at staging. Then run `git log --oneline -1` again and confirm the hash changed from the one you captured at the start. A non-empty `git diff --stat` is not sufficient proof of completion: staged-but-uncommitted work isn't part of any commit and can be silently lost if the working tree is cleaned up before a commit lands. Only report the improvement as done once the hash has actually changed.
|
||||
**Commit verification.** Inside a git worktree: once the audit is clean, run `git add` and `git commit` — do not stop at staging. Re-run `git log --oneline -1` and confirm the hash changed from Step 1's. A non-empty `git diff --stat` is not proof: staged-but-uncommitted work is part of no commit and is silently lost if the tree is cleaned up. Report done only once the hash has changed. Outside a worktree (a skill under `~/.claude/skills/`, say) nothing is committable — report done on a clean audit, naming that as the reason.
|
||||
|
||||
@@ -10,11 +10,32 @@ name: SKILL_NAME
|
||||
# Examples: my-tool, data-analyzer, pdf-processor
|
||||
|
||||
description: >
|
||||
FILL IN: What does this skill do? State capabilities specifically
|
||||
Use when FILL IN: trigger.
|
||||
FILL IN: at most ONE capability clause, stated specifically
|
||||
(e.g. "parses and validates OpenAPI specs", not "helps with APIs").
|
||||
Use when FILL IN: when should an agent activate this skill?
|
||||
Include indirect triggers: even if the user doesn't mention X explicitly.
|
||||
Do not use when FILL IN: near-miss exclusions — remove this line if none apply.
|
||||
Not FILL IN: near-miss case -> FILL IN: real sibling skill.
|
||||
# Required. Preloaded into EVERY session whether or not the skill is invoked.
|
||||
# Exactly three parts, in this order: trigger clause, at most one capability
|
||||
# clause, boundary clause. Drop the boundary line if no near-miss skill exists.
|
||||
# Trigger clause: when should an agent activate this skill? Describe the user's
|
||||
# intent, not the skill's internal mechanics.
|
||||
# Budget: 250 characters target, 400 hard ceiling (counting this value only,
|
||||
# with YAML folding resolved). This scaffold sits at 214 — keep the fill-in
|
||||
# under the target rather than growing past it.
|
||||
# Boundary clauses may be plural: write one per genuine near-miss, and none
|
||||
# where no sibling could steal activations.
|
||||
# Never let a hyphenated skill name wrap across two lines of this folded block
|
||||
# — folding turns the break into a space and the routing target stops resolving.
|
||||
# Banned here: capability lists, output-format detail, composition notes,
|
||||
# implementation detail, and restating one trigger twice in two registers.
|
||||
# The boundary target must resolve to a real skill or agent — it is checked.
|
||||
# Add "even if the user doesn't mention X explicitly" ONLY when the user's
|
||||
# natural phrasing genuinely omits the domain word.
|
||||
|
||||
# disable-model-invocation: true
|
||||
# Optional. Hand-invoked skills only (reached solely by the user typing
|
||||
# /SKILL_NAME). With this set, replace the description above with ONE plain
|
||||
# human-facing sentence — no trigger clause, no boundary clause.
|
||||
|
||||
# license: MIT
|
||||
# Optional. License name (e.g. MIT, Apache-2.0) or relative path to a bundled
|
||||
@@ -42,29 +63,30 @@ description: >
|
||||
---
|
||||
|
||||
<!-- ============================================================
|
||||
SKILL BODY
|
||||
SKILL BODY — decision procedure ONLY
|
||||
|
||||
Include only what the agent lacks:
|
||||
- Domain conventions the agent cannot infer from general knowledge
|
||||
- Non-obvious sequences or ordering constraints
|
||||
- One default per decision point + one escape hatch (never a menu)
|
||||
- Gotchas — facts that defy reasonable assumptions
|
||||
Keep here: ordered steps, decision branches, gates, and which reference
|
||||
file to load when.
|
||||
|
||||
Omit:
|
||||
- Concepts the agent already knows
|
||||
- Exhaustive option lists
|
||||
- Steps the agent handles independently
|
||||
- Restatements of the description
|
||||
Move to references/: lookup tables, spec restatements, output schemas,
|
||||
templates, example blocks, rationale prose, and anything only one branch
|
||||
reaches. Wire each one with the literal conditional form
|
||||
"If <condition>, read `references/<file>.md`." — a generic
|
||||
"see references/ for details" is a lint error.
|
||||
|
||||
Size budget: under 500 lines / 5000 tokens.
|
||||
Move reference material to references/ and load it conditionally.
|
||||
Bundle repeated executable logic into scripts/.
|
||||
Budget: 600 words target, 900 hard ceiling, counting THIS BODY ONLY
|
||||
(everything after the closing --- above). Separate from the whole-file
|
||||
spec backstop of 2,770 words / 500 lines — do not conflate them.
|
||||
|
||||
Delete this comment block before shipping.
|
||||
============================================================ -->
|
||||
|
||||
<!-- OPTIONAL: Gotchas section — highest-value content. Place near the top.
|
||||
Add facts that defy reasonable assumptions or non-obvious constraints.
|
||||
<!-- OPTIONAL but high-value: Gotchas. Place near the top — a gotcha read
|
||||
after the mistake is worthless.
|
||||
|
||||
Each entry states a fact that CONTRADICTS a reasonable default:
|
||||
something the agent gets wrong by acting sensibly. Maximum 5 entries.
|
||||
An entry that paraphrases a step below it is a failure, not a gotcha.
|
||||
|
||||
## Gotchas
|
||||
|
||||
@@ -72,7 +94,22 @@ description: >
|
||||
- FILL IN: non-obvious naming discrepancy or hidden constraint
|
||||
-->
|
||||
|
||||
<!-- OPTIONAL: Multi-step workflow checklist.
|
||||
<!-- DISPATCH — MANDATORY when this skill has two or more mutually exclusive
|
||||
flows. Keep only the dispatch table plus the gates common to every
|
||||
branch in this body; give each flow its own self-contained
|
||||
references/ file. Delete this block for a single-flow skill.
|
||||
|
||||
## Step 1 — Dispatch
|
||||
|
||||
| Condition | Flow | Reference |
|
||||
|---|---|---|
|
||||
| FILL IN: condition | FILL IN: flow | `references/FILL IN.md` |
|
||||
| FILL IN: condition | FILL IN: flow | `references/FILL IN.md` |
|
||||
|
||||
Read only the reference matching the resolved flow — each is self-contained.
|
||||
-->
|
||||
|
||||
<!-- OPTIONAL: single-flow workflow checklist. Delete if the skill dispatches.
|
||||
|
||||
## Workflow
|
||||
|
||||
@@ -81,23 +118,12 @@ description: >
|
||||
- [ ] Step 3: FILL IN
|
||||
-->
|
||||
|
||||
<!-- OPTIONAL: Output format template — use when the agent must produce a specific format.
|
||||
<!-- OPTIONAL: gates that apply to every branch — validation, versioning,
|
||||
closing checks. Keep these in the body even when flows are dispatched.
|
||||
|
||||
## Output format
|
||||
## Step N — Validate and close
|
||||
|
||||
Use this structure:
|
||||
|
||||
```markdown
|
||||
# [FILL IN: Title]
|
||||
|
||||
## FILL IN: Section
|
||||
FILL IN: what goes here
|
||||
```
|
||||
-->
|
||||
|
||||
<!-- OPTIONAL: Conditional reference — load documentation only when needed.
|
||||
|
||||
If FILL IN: condition, read `references/FILL IN: filename.md`.
|
||||
FILL IN: the check that must pass before this skill reports done.
|
||||
-->
|
||||
|
||||
## FILL IN: <section-name (e.g. Step 1, Workflow, Instructions)>
|
||||
|
||||
@@ -5,9 +5,17 @@ without bloating its core context.
|
||||
|
||||
## When to add a reference file
|
||||
|
||||
Move content here when SKILL.md is approaching 500 lines, or when a topic
|
||||
is only relevant in specific circumstances (error handling, edge cases,
|
||||
domain-specific sub-procedures).
|
||||
The SKILL.md body carries the decision procedure only. Everything else lives
|
||||
here: lookup tables, spec restatements, output schemas, templates, example
|
||||
blocks, rationale prose, and anything only one branch reaches.
|
||||
|
||||
Two triggers make a reference file mandatory rather than optional:
|
||||
|
||||
- The body is over its 600-word target (900 is a hard failure), counting the
|
||||
body only — everything after the frontmatter's closing `---`.
|
||||
- The skill has two or more mutually exclusive flows. The body then keeps only
|
||||
a dispatch table plus the gates common to every branch, and each flow gets
|
||||
its own self-contained file here (e.g. `create.md`, `improve.md`).
|
||||
|
||||
## How to reference from SKILL.md
|
||||
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
---
|
||||
source_keys:
|
||||
- agentskills-spec
|
||||
- agentskills-best-practices
|
||||
- agentskills-optimizing-descriptions
|
||||
---
|
||||
|
||||
# The description and body contract
|
||||
|
||||
House contract, set by ADR-0020. Every rule here is enforced by `/skill-audit` —
|
||||
`scripts/validate.sh` for the counts and the boundary targets, the bundled Vale styles for the
|
||||
prose patterns, and its reference files for the judgment calls.
|
||||
|
||||
## Why the budget exists
|
||||
|
||||
A skill's `name` and `description` are loaded into every agent's context at the start of every
|
||||
session, whether or not the skill is ever invoked. The body is loaded only on invocation, and then
|
||||
competes with the caller's live conversation. Those are two different costs, so they get two
|
||||
different ceilings — and a fat description is not merely expensive. A description that summarizes
|
||||
the workflow gets followed *instead of* the body: a description saying "code review between tasks"
|
||||
produced one review from a skill whose flowchart specified two.
|
||||
|
||||
## Description
|
||||
|
||||
A description carries exactly three things:
|
||||
|
||||
1. **Trigger clause** — when to invoke, imperative: "Use when ...", never "This skill ...".
|
||||
Focus on user intent, not the skill's internal mechanics.
|
||||
2. **At most one capability clause** — what it does, one clause, no enumeration. Be specific
|
||||
("parses and validates OpenAPI specs", not "helps with APIs").
|
||||
3. **Boundary clause** — form: `Not <thing> -> <skill-name>.` Add one only where a near-miss skill
|
||||
could steal activations.
|
||||
|
||||
Banned from a description; move it to the body or to `README.md`:
|
||||
|
||||
- Capability enumeration or feature lists
|
||||
- Output-format detail ("Produces a compact findings report with Why and Fix per finding")
|
||||
- Composition or architecture notes ("composes X rather than duplicating Y", "This is a
|
||||
cross-cutting shared skill", "the human-facing entry point")
|
||||
- Implementation detail ("Self-validates via a bundled deterministic script")
|
||||
- Restating the same trigger twice in two registers — a verb list, then the same verbs re-quoted
|
||||
as user phrasings. This is a FAIL, not a suggestion.
|
||||
|
||||
**Indirect triggers are conditional, not mandatory.** Add "even if the user doesn't mention X
|
||||
explicitly" only where the user's natural phrasing genuinely omits the domain word — true for the
|
||||
`gitea-*` family, because people say "create an issue" rather than "create a Gitea issue"; false
|
||||
for `git-commits`, where the user says "commit". Adding one everywhere is what inflated this
|
||||
corpus, and it was deleted as a blanket rule.
|
||||
|
||||
**Boundary targets must resolve.** Both forms are checked — the arrow and the prose form ("do not
|
||||
use for X, use `y` instead") — so a typo dangles either way. Targets resolve against a universe
|
||||
built by walking up **from the SKILL.md itself**: the nearest ancestor holding
|
||||
`plugins/*/.apm/{skills,agents}` (or, failing that, the nearest ancestor holding `.git`) contributes
|
||||
every skill and agent under `<root>/plugins/*/`, plus the skill's own apm package and the packages
|
||||
that package declares in `apm.yml` under `dependencies.apm`. A sibling plugin in the same monorepo
|
||||
therefore resolves; a skill in an unrelated repo does not. A boundary clause naming a target
|
||||
outside that universe sends the router nowhere and fails the audit. Check the target exists before
|
||||
writing it — do not invent a plausible sibling name.
|
||||
|
||||
That universe is the apm marketplace and stops there. A **host built-in is not a routing target**:
|
||||
`/compact`, `/clear` and `/init` are Claude Code slash commands with no counterpart in Copilot CLI
|
||||
or Codex, and `.apm/` source compiles for all three, so routing to one is a portability defect. The
|
||||
gate is right to fail it and there is no allowlist. If a built-in genuinely needs mentioning, write
|
||||
it un-slashed — ``the `compact` built-in`` — which makes no routing claim and is not checked.
|
||||
|
||||
**Length.** 250 characters SUGGESTION, 400 characters FAIL, counting the frontmatter value only
|
||||
with YAML folding resolved. The agentskills.io 1,024-character spec limit is unchanged and sits
|
||||
above both. The SUGGESTION tier is the one that moves the average; treat 250 as the target and 400
|
||||
as the outlier stop.
|
||||
|
||||
**Hand-invoked skills are exempt.** A skill carrying `disable-model-invocation: true` is absent
|
||||
from the model-visible listing and is reached only by the user typing `/name`. It takes one plain
|
||||
human-facing sentence — no trigger clause, no boundary clause, no indirect triggers. Worked
|
||||
example — the whole description of the `zoom-out` skill, which carries `disable-model-invocation`:
|
||||
|
||||
````markdown
|
||||
Tell the agent to zoom out and give broader context or a higher-level perspective. Use when
|
||||
you're unfamiliar with a section of code or need to understand how it fits into the bigger
|
||||
picture.
|
||||
````
|
||||
|
||||
## Body
|
||||
|
||||
The body carries the **decision procedure only**: ordered steps, decision branches, gates, and
|
||||
which reference to load when. Everything else moves to `references/`.
|
||||
|
||||
Ask of every sentence: "Would the agent get this wrong without it?" Cut anything that answers "no."
|
||||
|
||||
Include:
|
||||
|
||||
- Non-obvious sequences or ordering constraints — the agent may skip or reorder steps without this
|
||||
- Domain conventions the agent cannot infer from general knowledge — the core value a skill adds
|
||||
- One default per decision point, plus one escape hatch — never a menu; menus cause the agent to
|
||||
pause or pick arbitrarily
|
||||
- Gotchas — facts that defy reasonable assumptions
|
||||
|
||||
Exclude:
|
||||
|
||||
- Concepts the agent already knows (what JSON is, how HTTP works) — tokens without behavior change
|
||||
- Exhaustive option lists — pick a default; the agent does not benefit from choosing
|
||||
- Steps the agent handles independently — over-specifying leads agents down unproductive paths
|
||||
- Restatements of the description — it is already in context
|
||||
|
||||
Move to `references/`: lookup tables, spec restatements, output schemas, templates, example
|
||||
blocks, rationale prose, and any content only one branch reaches. Each reference file is
|
||||
self-contained for its concern, and every one is wired from the body with the literal conditional
|
||||
form:
|
||||
|
||||
**The one exception, stated once so it is not re-litigated:** an output schema stays in the body
|
||||
only when it applies to *every* flow and is short — roughly 50 words or less, which is the "Output
|
||||
format template" pattern below. An output schema that is longer than that, or that only one flow
|
||||
produces, moves to `references/` like any other schema. No third option exists, and the two rules
|
||||
do not disagree.
|
||||
|
||||
````markdown
|
||||
If <condition>, read `references/<file>.md`.
|
||||
````
|
||||
|
||||
A generic pointer ("see references/ for details") is a Vale error — the agent cannot act on it.
|
||||
|
||||
**Dispatch is mandatory at two or more mutually exclusive flows.** The body carries the dispatch
|
||||
table and the gates common to every branch; each flow gets its own self-contained `references/`
|
||||
file. Exemplar: the `apm-workflow` skill — a **421-word body** dispatching to 3,006 words of
|
||||
references. Calibrate against 421: that file's whole-file count is 554 words, and aiming at that
|
||||
number instead overshoots the body budget by ~30%.
|
||||
|
||||
**Length.** 600 words SUGGESTION, 900 words FAIL, counting the **body only** — everything after
|
||||
the frontmatter's closing `---`.
|
||||
|
||||
## Gotchas section
|
||||
|
||||
- Each entry must state a fact that **contradicts a reasonable default** — something the agent
|
||||
gets wrong by acting sensibly. "Never commit secrets" is not one; the agent already knows.
|
||||
- More than five entries is a SUGGESTION — five is the guideline, not a ceiling.
|
||||
- A Gotcha that paraphrases a step in the body below it is a **FAIL**. If the rule is already a
|
||||
step, it is not a gotcha.
|
||||
- A Gotchas section exceeding 25% of the body is a SUGGESTION.
|
||||
- Place the section near the top — a gotcha read after the mistake is worthless.
|
||||
|
||||
## Two size gates, two measurements
|
||||
|
||||
| Gate | SUGGESTION | FAIL | Counts |
|
||||
|---|---|---|---|
|
||||
| description | 250 chars | 400 chars | the `description:` value only |
|
||||
| body | 600 words | 900 words | the body only, after the closing `---` |
|
||||
| spec backstop | — | 1,024 chars | the `description:` value only |
|
||||
| spec backstop | — | 2,770 words / 500 lines | the **whole file**, frontmatter included |
|
||||
|
||||
The 600/900 pair and the 2,770/500 pair are not the same measurement and must not be unified: the
|
||||
first is a quality gate on what the caller's context absorbs, the second a conformance backstop on
|
||||
the file. A skill can sit well inside one and fail the other.
|
||||
|
||||
When a body approaches its ceiling, relocate rather than delete — move reference material to
|
||||
`references/<topic>.md` behind a conditional trigger, and bundle repeated executable logic into
|
||||
`scripts/` rather than reinventing it each run.
|
||||
|
||||
## Body patterns
|
||||
|
||||
**Default with escape hatch** (not a menu):
|
||||
|
||||
````markdown
|
||||
Use <X> for <task>. For <edge case>, use <Y> instead.
|
||||
````
|
||||
|
||||
**Prescriptive sequence** (when order is critical or fragile):
|
||||
|
||||
````markdown
|
||||
Run exactly:
|
||||
```bash
|
||||
<command>
|
||||
```
|
||||
Do not modify flags.
|
||||
````
|
||||
|
||||
**Checklist** (multi-step workflows):
|
||||
|
||||
````markdown
|
||||
- [ ] Step 1: ...
|
||||
- [ ] Step 2: ...
|
||||
````
|
||||
|
||||
**Dispatch table** (two or more mutually exclusive flows):
|
||||
|
||||
````markdown
|
||||
| Condition | Flow | Reference |
|
||||
|---|---|---|
|
||||
| <condition> | <flow> | `references/<file>.md` |
|
||||
````
|
||||
|
||||
**Output format template** (when the skill produces structured output on *every* flow, and the
|
||||
schema is roughly 50 words or less — see the exception under Body above; anything longer or
|
||||
flow-specific belongs in `references/`):
|
||||
|
||||
````markdown
|
||||
Output format:
|
||||
```
|
||||
<field>: <value>
|
||||
```
|
||||
````
|
||||
|
||||
For longer templates, place them in `references/<topic>.md` or `assets/<name>.md` and reference
|
||||
conditionally.
|
||||
|
||||
## Embedding org-specific policy
|
||||
|
||||
If a skill encodes a rule sourced from an org convention file (e.g. `core/instructions/*.md`),
|
||||
inline that content directly into the skill (SKILL.md or a `references/` file) rather than pointing
|
||||
to the file's path. Plugins must be self-contained and portable — the org file may not exist
|
||||
wherever the plugin is installed, and in this repo such files are meant to be deleted once their
|
||||
content is fully embedded downstream. Tag the inlined content with a `source_keys` entry using the
|
||||
same `references/sources.md` schema as the create flow's Step 6, noting in the `Research doc:`
|
||||
field that the source is an org convention rather than a plugin research corpus entry, so
|
||||
provenance survives after the source file is gone.
|
||||
179
plugins/kyberforge/.apm/skills/skill-author/references/create.md
Normal file
179
plugins/kyberforge/.apm/skills/skill-author/references/create.md
Normal file
@@ -0,0 +1,179 @@
|
||||
---
|
||||
source_keys:
|
||||
- agentskills-home
|
||||
- agentskills-spec
|
||||
- agentskills-best-practices
|
||||
- agentskills-quickstart
|
||||
- agentskills-using-scripts
|
||||
---
|
||||
|
||||
# Creating a new skill
|
||||
|
||||
Return to `SKILL.md` Step 4 once Step 6 below is done — validation, versioning and commit
|
||||
verification are shared with the improve flow and are not repeated here.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Run `/grill-me` on the skill's design and research the target domain first. Share those outputs
|
||||
in this conversation: grill context, research docs, examples, constraints.
|
||||
|
||||
Design for one coherent user intent — skills too narrow force multiple loads per task; too broad
|
||||
are hard to activate precisely.
|
||||
|
||||
Before touching the filesystem, verify you have:
|
||||
|
||||
- [ ] A clear purpose — what specific task will this skill handle?
|
||||
- [ ] Trigger scenarios — when should an agent activate it?
|
||||
- [ ] Skill name (kebab-case) and destination path
|
||||
|
||||
If any are missing, stop and ask the user before proceeding.
|
||||
|
||||
**Requires `/skill-audit`** — used in `SKILL.md` Step 4 for final validation. Both skills ship in
|
||||
the kyberforge plugin and are co-installed. If `/skill-audit` is unavailable, stop and ask the
|
||||
user to install the kyberforge plugin before continuing.
|
||||
|
||||
## Package-intent gate
|
||||
|
||||
Judge whether the destination is meant to be inside an APM package before running the scaffold
|
||||
script — the script cannot tell "no package here" apart from "package not scaffolded yet":
|
||||
|
||||
- Package intent but no `type:`-bearing `apm.yml` found at or above the destination (e.g. "add to
|
||||
my apm package", or a sibling `.apm/`/`apm.yml` exists nearby) → **stop**, tell the user to run
|
||||
`/apm-workflow configure` (`apm plugin init`, from inside the package directory) first, then
|
||||
retry. Do not fall through to standalone mode.
|
||||
- Otherwise (a `~/`-rooted destination, or no package context implied) → continue to Step 1.
|
||||
|
||||
## Step 1 — Scaffold
|
||||
|
||||
Run the copy script with the skill name and a path inside or at the target:
|
||||
|
||||
```bash
|
||||
bash scripts/new-skill.sh <skill-name> <path>
|
||||
```
|
||||
|
||||
The script walks up from `<path>` for a package boundary: an ancestor `apm.yml` with a top-level
|
||||
`type:` field (`instructions`/`skill`/`hybrid`/`prompts`) means **package mode** — scaffolds into
|
||||
`<package-root>/.apm/skills/<skill-name>/`, not under `<path>` (a subdirectory of the package
|
||||
works fine as `<path>`). A `type:`-less `apm.yml` is a marketplace-only manifest, skipped. Hitting
|
||||
`.git` or the filesystem root first means **standalone mode** — scaffolds directly into
|
||||
`<path>/<skill-name>/`.
|
||||
|
||||
Examples:
|
||||
|
||||
```bash
|
||||
# Package mode — packages/my-pkg/apm.yml already has `type: skill`
|
||||
bash scripts/new-skill.sh my-tool packages/my-pkg/
|
||||
|
||||
# Standalone mode — no apm.yml/.git above ~/.agents/skills/
|
||||
bash scripts/new-skill.sh my-tool ~/.agents/skills/
|
||||
```
|
||||
|
||||
The script prints which mode it used and where the skill landed — read its output.
|
||||
|
||||
In package mode, read `references/deployment-modes.md` before adding any file references to
|
||||
SKILL.md.
|
||||
|
||||
## Step 2 — Update `apm.yml` includes (package mode only)
|
||||
|
||||
Skip in standalone mode. In package mode, check the resolved package's `apm.yml`: if `includes:`
|
||||
is an explicit list (not `auto`), append `.apm/skills/<skill-name>/` to it if not already present,
|
||||
preserving YAML formatting. If `includes: auto` or the field is absent, do nothing — `auto`
|
||||
already covers the new skill. Use Read/Edit directly on `apm.yml`; this is not part of
|
||||
`scripts/new-skill.sh`.
|
||||
|
||||
## Step 3 — Fill in SKILL.md
|
||||
|
||||
Open the new skill's `SKILL.md` (the path Step 1 printed) and replace every `FILL IN:`
|
||||
placeholder. The scaffold template already carries the compliant frontmatter and body skeleton —
|
||||
fill it rather than restructuring it.
|
||||
|
||||
**`name`** — already set by the scaffold script. Must exactly match the directory name. Format:
|
||||
1–64 characters, lowercase letters, numbers and hyphens only; no leading, trailing or consecutive
|
||||
hyphens (`--`).
|
||||
|
||||
**`description`** — carries the entire triggering burden and is preloaded every session. Write it
|
||||
against `references/contract.md`, which holds the three-part shape, the banned content, the
|
||||
boundary-clause form and the length tiers. A hand-invoked skill (`SKILL.md` Step 2) takes one
|
||||
plain sentence and `disable-model-invocation: true` instead.
|
||||
|
||||
**Optional frontmatter** — uncomment and fill in, or remove entirely:
|
||||
|
||||
- `license` — include when distributing the skill externally
|
||||
- `compatibility` — include if the skill requires specific tools, runtimes, or network access
|
||||
(max 500 characters)
|
||||
- `metadata` — key-value map; use `author`, `version`, `category`; add `source_keys` now (Step 6)
|
||||
if research sources are in context
|
||||
- `allowed-tools` — space-separated pre-approved tools; reduces permission prompts (experimental —
|
||||
support varies by client)
|
||||
- `disable-model-invocation` — hand-invoked skills only
|
||||
|
||||
**`metadata.source_keys`** — if research sources are in context, list the relevant slugs as you
|
||||
write the body; do not defer this to Step 6. Agents that fill in `source_keys` late tend to omit
|
||||
it entirely. Example:
|
||||
|
||||
```yaml
|
||||
metadata:
|
||||
source_keys:
|
||||
- my-source-slug
|
||||
- another-slug
|
||||
```
|
||||
|
||||
**Body** — write the decision procedure only, following the body rules and patterns in
|
||||
`references/contract.md`. Rename the placeholder section headings to ones that fit the skill's
|
||||
structure.
|
||||
|
||||
## Step 4 — Add scripts (if needed)
|
||||
|
||||
Place executable scripts in `scripts/`. Critical rule: **no interactive prompts** — agents run
|
||||
non-interactive, and blocking on TTY input hangs indefinitely. Accept all input via flags, env
|
||||
vars, or stdin.
|
||||
|
||||
If adding a script, read `references/scripts.md` first — it covers the full contract: structured
|
||||
output, pinned versions, self-contained deps, idempotency, exit codes, dry-run, error messages,
|
||||
and output size limits.
|
||||
|
||||
If no scripts are needed, delete `scripts/README.md` and the `scripts/` directory.
|
||||
|
||||
## Step 5 — Add references, assets, and tests (if needed)
|
||||
|
||||
**`references/`** — additional documentation loaded on demand. One topic per file, named in
|
||||
kebab-case after the topic. Reference conditionally from SKILL.md with the literal form
|
||||
``If <condition>, read `references/<file>.md` ``.
|
||||
|
||||
**Two hops from `SKILL.md`, never three.** A flow file may route on to a shared contract or
|
||||
sub-topic file — that is the shipped pattern here (`SKILL.md` → `references/create.md` → this
|
||||
file's own pointers to `contract.md`, `scripts.md` and `deployment-modes.md`). What does not work
|
||||
is a third hop: a file reachable only through two intermediates is rarely loaded at the moment it
|
||||
is needed. Every hop past the first also needs the same literal conditional form, so the agent
|
||||
knows when to take it.
|
||||
|
||||
**`assets/`** — static resources: templates, schemas, lookup tables. Reference by relative path
|
||||
from SKILL.md.
|
||||
|
||||
**`tests/`** — test files for scripts in `scripts/`. Use when scripts are complex enough to break
|
||||
silently. Test infrastructure (`.bats`, `*_test.*`) belongs here, not in `scripts/`. See
|
||||
`tests/README.md` for setup instructions.
|
||||
|
||||
If not needed, delete the placeholder READMEs and their directories.
|
||||
|
||||
## Step 6 — Populate or delete `references/sources.md`
|
||||
|
||||
If a research `sources.md` is present in the conversation context:
|
||||
|
||||
1. Read it and filter to entries with `` `extracted` `` status only.
|
||||
2. For each entry, determine which skill files it contributed to (SKILL.md and any files in
|
||||
`references/` that drew from it). Update `Contributing files` accordingly — list skill files,
|
||||
not research topic files.
|
||||
3. Write the updated content to `references/sources.md`. For each entry, include
|
||||
`- **Research doc:** <path>` where `<path>` is the relative path from the repo root to the
|
||||
plugin-level research sources file this entry was drawn from (e.g.
|
||||
`plugins/myplugin/docs/research/docs/<topic>/sources.md`). This field is required on every
|
||||
entry — it makes the provenance chain explicit and is validated by `/skill-audit`.
|
||||
4. Add `source_keys` to the frontmatter of `SKILL.md` (under `metadata`) listing the slugs of
|
||||
sources that informed it.
|
||||
5. For each file in `references/` that was informed by research sources, add `source_keys`
|
||||
frontmatter (same format as research topic files) listing the relevant slugs.
|
||||
|
||||
If no research `sources.md` is in context, delete `references/sources.md`.
|
||||
|
||||
Then return to `SKILL.md` Step 4.
|
||||
@@ -0,0 +1,94 @@
|
||||
---
|
||||
source_keys:
|
||||
- agentskills-best-practices
|
||||
- agentskills-evaluating-skills
|
||||
- agentskills-optimizing-descriptions
|
||||
---
|
||||
|
||||
# Improving an existing skill
|
||||
|
||||
Return to `SKILL.md` Step 4 once Step 4 below is done — validation, versioning and commit
|
||||
verification are shared with the create flow and are not repeated here.
|
||||
|
||||
## Step 1 — Verify inputs
|
||||
|
||||
Confirm the skill directory path exists and that at least one improvement signal is present in the
|
||||
conversation or a referenced file.
|
||||
|
||||
If the skill directory is missing, ask for it. If no signals are present, stop: "This skill applies
|
||||
existing signals to a skill. For a blind review without signals, use `/skill-audit` instead."
|
||||
|
||||
Signals can come from anywhere in the conversation or referenced files:
|
||||
|
||||
- Grill session output (most common predecessor in the factory sequence)
|
||||
- `/skill-audit` findings (PASS/FAIL/SUGGESTION punch list)
|
||||
- Human feedback (feedback.json, inline in conversation, PR or issue comments)
|
||||
- Session context describing what went wrong
|
||||
|
||||
Also verify the `name` field in frontmatter matches the skill's directory name exactly.
|
||||
|
||||
## Step 2 — Gather and group signals
|
||||
|
||||
Read the current skill files (SKILL.md and any files in `scripts/`, `references/`, `assets/`,
|
||||
`tests/`). Then collect all signals from the conversation and any file paths the user has
|
||||
referenced.
|
||||
|
||||
Group signals by **root cause**, not symptom. Patching per symptom is the default failure mode:
|
||||
three eval failures may all trace to one missing instruction. Ask: "What single gap in the skill
|
||||
causes this cluster of failures?" One root cause → one fix. Do not make a separate edit for each
|
||||
symptom.
|
||||
|
||||
```text
|
||||
Example:
|
||||
- Session context: output format is wrong on every run
|
||||
- Audit finding: no output template defined
|
||||
- User feedback: "I always have to ask it to format the output"
|
||||
→ Root cause: SKILL.md has no output format specification → one fix: add an output template
|
||||
```
|
||||
|
||||
## Step 3 — Announce planned changes
|
||||
|
||||
Before editing, state:
|
||||
|
||||
- Which root causes were identified and what evidence supports each
|
||||
- Which files will be changed and what will change in each
|
||||
|
||||
Then proceed — edits are reversible via git, no approval checkpoint needed.
|
||||
|
||||
## Step 4 — Apply changes
|
||||
|
||||
Edit any file in the skill directory that the signals point to: SKILL.md, `scripts/`,
|
||||
`references/`, `assets/`, `tests/`, README.md.
|
||||
|
||||
**Generalize, do not patch.** Find the underlying gap, not the specific example that failed. A fix
|
||||
scoped only to the test cases you have seen will overfit and perform worse on new inputs.
|
||||
|
||||
**Keep it lean.** Remove instructions that are not pulling their weight. For every sentence you
|
||||
add, ask: "Would the agent get this wrong without it?" A shorter, focused skill consistently
|
||||
outperforms an exhaustive one.
|
||||
|
||||
**Explain the why.** Reasoning-based instructions outperform rigid directives. If you find yourself
|
||||
writing a rule in all caps (ALWAYS/NEVER), reframe it: explain why the behavior matters so the
|
||||
agent can apply judgment in edge cases.
|
||||
|
||||
**Retrofit before extending.** Any edit to a skill that predates ADR-0020 has to bring it into the
|
||||
contract first — the gates are hot and carry no baseline file, so a one-line fix to a
|
||||
non-compliant skill cannot be committed until the description and body meet
|
||||
`references/contract.md`. Treat that retrofit as part of the same change, not a follow-up.
|
||||
|
||||
If the skill's description exceeds 250 characters, or its body-only word count exceeds 600, read
|
||||
`references/retrofit.md` before editing. It carries the ordered cut procedure, the
|
||||
mutually-exclusive-flows test, the reference-file conventions this flow needs, the collateral
|
||||
checklist for `README.md` and `references/sources.md`, and a worked description retrofit. Do not
|
||||
improvise the cuts — four dry runs invented six to ten different answers to the same questions.
|
||||
|
||||
If a signal points to a script or reference file, edit that file directly rather than adding a
|
||||
workaround in SKILL.md.
|
||||
|
||||
**Check for regressions before handing back.** `SKILL.md` Step 4 tells you to resolve every FAIL,
|
||||
which says nothing about a check that passed *before* these edits and no longer does. Compare the
|
||||
closing audit against the skill's pre-edit state — a PASS that has become a SUGGESTION, or a
|
||||
SUGGESTION that has become a FAIL, is damage this flow caused and is in scope for it. Only the
|
||||
improve flow can make that comparison; the create flow has no prior state to compare against.
|
||||
|
||||
Then return to `SKILL.md` Step 4.
|
||||
@@ -0,0 +1,152 @@
|
||||
---
|
||||
source_keys:
|
||||
- agentskills-best-practices
|
||||
- agentskills-optimizing-descriptions
|
||||
---
|
||||
|
||||
# Retrofitting a skill to the ADR-0020 contract
|
||||
|
||||
Read this when `references/improve.md` Step 4 sends you here: the skill you are editing is over
|
||||
the description or body budget and has to come into contract before any other change can be
|
||||
committed. The gates are hot and carry no baseline file, so a one-line fix to a non-compliant
|
||||
skill is blocked until this is done.
|
||||
|
||||
Measure first. Do not guess which gate fired: run `/skill-audit` on the directory and read its
|
||||
`### Structure` dimension, which reports the description characters and the **body-only** word
|
||||
count separately from the whole-file spec backstop. Retrofit against the number that actually
|
||||
fired — a skill can sit a thousand words inside the whole-file backstop while failing the body
|
||||
budget.
|
||||
|
||||
**Validate in place.** Audit the skill's real directory inside its package. Never audit a copy in a
|
||||
scratch directory, and never move a skill out to work on it: the boundary-target universe is built
|
||||
by walking up *from the file being checked*, so a copy with no authoring root above it resolves
|
||||
against nothing and the check declines rather than running —
|
||||
|
||||
```text
|
||||
INFO boundary-target resolution DID NOT RUN — no skill universe could be determined for
|
||||
this path ... Unchecked target(s): totally-fake-target
|
||||
```
|
||||
|
||||
The run still exits 0, so that line reads as a pass and is not one. Treat `DID NOT RUN` as **not
|
||||
checked**, always. A retrofit signed off on a scratch copy carries an unverified boundary target
|
||||
into the corpus, which is precisely the failure this gate exists to catch.
|
||||
|
||||
## Cut in this order
|
||||
|
||||
Work the list top down and stop as soon as the gate clears. The order is by ratio of tokens
|
||||
removed to behaviour lost — inverting it is how a retrofit ends up deleting the one instruction
|
||||
the skill existed to carry.
|
||||
|
||||
1. **Gotchas that paraphrase a step in the body below.** Zero information, and already a FAIL on
|
||||
its own. Delete the Gotcha, keep the step.
|
||||
2. **Spec restatements** — text that repeats a published specification, a tool's `--help`, or a
|
||||
ceiling the validator already enforces. The agent gets this right without it. Delete, or move
|
||||
the table to `references/` if a flow genuinely needs to look it up.
|
||||
3. **Capability enumeration** — in a description, the feature list after the trigger clause; in a
|
||||
body, the paragraph that recites what the skill can do. One capability clause survives in the
|
||||
description; the rest belongs in `README.md`.
|
||||
4. **Per-flow prose** — anything only one branch of the procedure ever reaches. This is the
|
||||
largest single win in most bodies, and it is a *move*, not a delete: each flow gets its own
|
||||
self-contained `references/` file, wired from a dispatch table.
|
||||
|
||||
If the body is still over after all four, the skill is doing two jobs. Split it, and say so
|
||||
rather than compressing prose until it stops being readable.
|
||||
|
||||
## What "mutually exclusive flows" means
|
||||
|
||||
Two or more flows that a single invocation cannot both take. The three-way test, copied verbatim
|
||||
from the body-discipline rubric `/skill-audit` judges against — nothing to load, it is quoted in
|
||||
full here:
|
||||
|
||||
> separate subcommands, separate input types, separate lifecycle stages
|
||||
|
||||
Any one of the three is enough. Two flows that differ only in a parameter value are one flow.
|
||||
At two or more mutually exclusive flows a dispatch table is **mandatory** regardless of word
|
||||
count, because every invocation otherwise pays for every branch it did not take.
|
||||
|
||||
## Reference-file conventions
|
||||
|
||||
The create flow owns these rules, and this flow is forbidden from reading `references/create.md`,
|
||||
so what a retrofit needs is restated here:
|
||||
|
||||
- **One topic per file.** A file mixing two concerns gets loaded for one of them and spends the
|
||||
caller's context on the other.
|
||||
- **Kebab-case filenames**, named after the topic rather than the flow that reads it —
|
||||
`body-discipline.md`, not `step-3.md`.
|
||||
- **Wire every file with the literal conditional form** ``If <condition>, read
|
||||
`references/<file>.md` ``. A generic pointer ("see `references/` for details") is a Vale error.
|
||||
- **Two hops from `SKILL.md`, never three.** A flow file may route on to a shared contract file;
|
||||
a file reachable only through two intermediates is rarely loaded when it is needed.
|
||||
- **`source_keys` frontmatter.** If the content you are moving drew on a research source, the new
|
||||
file needs top-level `source_keys:` frontmatter listing those slugs, and every slug must already
|
||||
exist as an `## <slug>` heading in `references/sources.md`. Moving sourced content out of
|
||||
`SKILL.md` without carrying its slugs across breaks the provenance chain, and `/skill-audit`
|
||||
reports the new file as an INFO with no `source_keys`.
|
||||
|
||||
## Collateral is mandatory, not optional
|
||||
|
||||
Moving content out of a `SKILL.md` leaves three files describing a structure that no longer
|
||||
exists. `/skill-audit`'s provenance check exits clean on all three of these, so nothing catches
|
||||
them for you. After every retrofit that adds, removes or renames a file:
|
||||
|
||||
- [ ] **`README.md` file table** — a row for every new `references/` file, and no row left for a
|
||||
file that is gone. Say what triggers the load, not just what the file contains.
|
||||
- [ ] **`references/README.md`**, where the skill has one — same update, same reason.
|
||||
- [ ] **`references/sources.md` → `Contributing files`** — add the new file to every slug whose
|
||||
content moved into it, and remove any file the retrofit deleted. This is the one that gets
|
||||
missed: `sources.md` keeps citing sections of `SKILL.md` that no longer exist, the
|
||||
provenance check still exits 0, and the stale claim survives review.
|
||||
- [ ] Re-run `/skill-audit` and confirm its `### Provenance` dimension does not report the new
|
||||
file as missing `source_keys`.
|
||||
|
||||
## Worked example — a description retrofit
|
||||
|
||||
`gitea-issues` before, 827 characters, the single most common shape in the corpus:
|
||||
|
||||
```text
|
||||
Use when reading or writing Gitea issues: listing repo issues, getting a single issue's details/
|
||||
comments/labels, creating an issue, updating its state, adding or editing comments, applying
|
||||
labels via issue_write, or searching issues/PRs across repositories. Triggers on "create an
|
||||
issue", "what issues are open", "get issue #N", "close issue #N", "comment on issue #N", "search
|
||||
issues for X" — even when the user doesn't say "Gitea" explicitly. Composes gitea-labels-
|
||||
milestones for all label inference/resolution and milestone lookup — do not use this skill to
|
||||
manage label or milestone definitions themselves (create/edit/delete a label, create/close a
|
||||
milestone), that's gitea-labels-milestones directly. Do not use for pull requests (use gitea-prs)
|
||||
or for local git branch/commit work (use gitea-branches or git-branches).
|
||||
```
|
||||
|
||||
After, 240 characters:
|
||||
|
||||
```text
|
||||
Use when reading or writing Gitea issues — list, read, create, comment on, label, close, or
|
||||
search — even when the user does not say "Gitea". Not pull requests -> `gitea-prs`. Not label or
|
||||
milestone definitions -> `gitea-labels-milestones`.
|
||||
```
|
||||
|
||||
What came out, and why:
|
||||
|
||||
| Removed | Why |
|
||||
|---|---|
|
||||
| The second trigger register — `Triggers on "create an issue", "what issues are open", …` | The same triggers restated as quoted user phrasings. Two registers of one trigger list is a FAIL, not a suggestion. |
|
||||
| `applying labels via issue_write` | Implementation detail. The router does not choose a skill by which MCP call it makes. |
|
||||
| `Composes gitea-labels-milestones for all label inference/resolution and milestone lookup` | A composition note. It changes no routing decision and belongs in `README.md`. |
|
||||
| The parenthetical `(create/edit/delete a label, create/close a milestone)` | Capability enumeration inside a boundary clause. The boundary needs the target, not its feature list. |
|
||||
| The `gitea-branches` / `git-branches` boundary | Dropped entirely. Neither was ever going to win an issue request, so the clause defended against nothing — an invented boundary costs characters and buys no routing accuracy. |
|
||||
| `Do not use for pull requests (use gitea-prs)` prose form | Kept, but rewritten as `Not pull requests -> \`gitea-prs\`.` The rewrite buys characters and one uniform shape for the router — not safety. Both forms are parsed **and** target-checked, so a typo in the prose form dangles exactly as an arrow typo does. |
|
||||
|
||||
What stayed: one trigger clause, one capability clause, the indirect trigger (genuinely warranted
|
||||
here — people say "create an issue", not "create a Gitea issue"), and the boundary clauses.
|
||||
|
||||
## Two rules the gates enforce but the prose does not spell out
|
||||
|
||||
**Boundary clauses may be plural.** Write one per genuine near-miss — the example above carries
|
||||
two, because two different skills could each steal activations. "A boundary clause" in the
|
||||
contract means *at least one*, not *exactly one*. What is banned is a boundary clause invented for
|
||||
a skill that was never going to compete, not a second real one.
|
||||
|
||||
**Never let a hyphenated routing target wrap across lines in a folded `>` scalar.** YAML folding
|
||||
replaces the newline with a space, so `gitea-labels-` at the end of one line and `milestones` at
|
||||
the start of the next fold into `gitea-labels- milestones`. `validate.sh` then reads the target as
|
||||
`gitea-labels`, finds no such skill, and reports a dangling boundary target — the live finding on
|
||||
`gitea-issues` today. Reflow the line so the whole name sits on one of them. The same applies to
|
||||
any backticked skill or agent name in a description.
|
||||
@@ -18,7 +18,7 @@ source_keys:
|
||||
- **URL:** https://agentskills.io/home.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** Agent Skills overview — what it is, why it exists, progressive disclosure model, ecosystem of 35+ implementing tools
|
||||
- **Contributing files:** SKILL.md
|
||||
- **Contributing files:** SKILL.md, references/create.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## agentskills-spec
|
||||
@@ -26,7 +26,7 @@ source_keys:
|
||||
- **URL:** https://agentskills.io/specification.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** Complete SKILL.md format specification — frontmatter fields, constraints, body content, optional directories, progressive disclosure levels, file references, validation
|
||||
- **Contributing files:** SKILL.md, references/deployment-modes.md
|
||||
- **Contributing files:** SKILL.md, references/create.md, references/contract.md, references/deployment-modes.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## agentskills-best-practices
|
||||
@@ -34,7 +34,7 @@ source_keys:
|
||||
- **URL:** https://agentskills.io/skill-creation/best-practices.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** Best practices for skill creators — starting from real expertise, spending context wisely, calibrating control, instruction patterns (gotchas, templates, checklists, validation loops)
|
||||
- **Contributing files:** SKILL.md
|
||||
- **Contributing files:** SKILL.md, references/create.md, references/improve.md, references/contract.md, references/retrofit.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## agentskills-optimizing-descriptions
|
||||
@@ -42,7 +42,7 @@ source_keys:
|
||||
- **URL:** https://agentskills.io/skill-creation/optimizing-descriptions.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** How to systematically test and improve skill descriptions for triggering accuracy — eval queries, trigger rate testing, train/validation splits, optimization loop
|
||||
- **Contributing files:** SKILL.md
|
||||
- **Contributing files:** SKILL.md, references/improve.md, references/contract.md, references/retrofit.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## agentskills-evaluating-skills
|
||||
@@ -50,7 +50,7 @@ source_keys:
|
||||
- **URL:** https://agentskills.io/skill-creation/evaluating-skills.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** Eval-driven skill quality improvement — test case design, workspace structure, assertion writing, grading, benchmarking, human review, iteration loop
|
||||
- **Contributing files:** SKILL.md
|
||||
- **Contributing files:** SKILL.md, references/improve.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## agentskills-using-scripts
|
||||
@@ -58,7 +58,7 @@ source_keys:
|
||||
- **URL:** https://agentskills.io/skill-creation/using-scripts.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** Using scripts in skills — one-off commands, self-contained scripts with inline dependencies, designing scripts for agentic use (no interactive prompts, --help, structured output, idempotency)
|
||||
- **Contributing files:** SKILL.md, references/scripts.md
|
||||
- **Contributing files:** SKILL.md, references/create.md, references/scripts.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
## agentskills-quickstart
|
||||
@@ -66,5 +66,5 @@ source_keys:
|
||||
- **URL:** https://agentskills.io/skill-creation/quickstart.md
|
||||
- **Research doc:** plugins/kyberforge/docs/research/docs/agentskillsio/sources.md
|
||||
- **Description:** Step-by-step guide to creating a first skill (roll-dice example), how discovery/activation/execution work in practice
|
||||
- **Contributing files:** SKILL.md
|
||||
- **Contributing files:** SKILL.md, references/create.md
|
||||
- **Status:** `extracted`
|
||||
|
||||
@@ -180,7 +180,8 @@ else
|
||||
fi
|
||||
echo "" >&2
|
||||
echo "Next steps:" >&2
|
||||
echo " 1. Fill in $TARGET/SKILL.md — replace all FILL IN: placeholders" >&2
|
||||
echo " 1. Fill in $TARGET/SKILL.md — replace all FILL IN: placeholders." >&2
|
||||
echo " Description: 250 chars target / 400 ceiling. Body: 600 / 900, body only." >&2
|
||||
echo " 2. Add scripts to scripts/ if needed (or delete the directory)" >&2
|
||||
echo " 3. Add docs to references/ if needed (or delete the directory)" >&2
|
||||
echo " 4. Add resources to assets/ if needed (or delete the directory)" >&2
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "kyberforge",
|
||||
"version": "1.4.1",
|
||||
"version": "1.6.0",
|
||||
"description": "Skills and agents for creating, maintaining, and managing a Claude Code / Copilot CLI plugin marketplace.",
|
||||
"author": {
|
||||
"name": "Defame1297",
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user