Files
holocron/plugins/kyberforge/.apm/skills/factory-audit/tests/validate-skill.bats
Defame1297 e849a823f7 fix(gates): report an unparsed routing clause beside a parsing sibling
boundary_clause_status() ran BOUNDARY_ARROW.search() and _arrow_targets()
over the whole description, so one arrow clause that parsed suppressed the
diagnostic for every other clause in it. A backticked hyphenated routing
target wrapped across lines in a folded scalar was therefore silently
unchecked -- no error, no suggestion, exit 0 -- whenever the description
carried one other clause that parsed. Written bare, the same wrap errors
correctly. That is the shape #100 regressed on.

The check is now per clause. Nothing that passed starts failing: all 68
routing targets across the 38 SKILL.md files resolved before and still do.
26 of those descriptions carry more than one arrow clause, so the
suppression was live across two thirds of the corpus, not an edge case.

validate-skill.bats pins the shape. test-adr0020-targets.sh's comment
described the #100 regression as a backticked wrap; the historical text was
unbackticked, which is precisely the shape the gate did not catch.

Also closes three README misroutes the branch left in the enforcement
layer: CompositionNote.yml's message, agent-description-quality.md:58 and
vale-wrap.sh's header still sent overflow to a skill-root README.md and
named the two skills ADR-0025 merged away. 1ec3e8a fixed the prose and
missed the rules that enforce it.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01NwD8Egs5r4ndqeFLmhusX2
2026-09-20 18:37:45 +00:00

1128 lines
44 KiB
Bash

#!/usr/bin/env bats
setup() {
REPO_ROOT="$(cd "$BATS_TEST_DIRNAME/../../../../../../" && pwd)"
load "$REPO_ROOT/tests/test_helper/bats-support/load"
load "$REPO_ROOT/tests/test_helper/bats-assert/load"
SCRIPT="$(cd "$BATS_TEST_DIRNAME/../scripts" && pwd)/validate.sh"
TMPDIR="$(mktemp -d)"
# Helper: create a minimal valid skill directory.
#
# The description carries a boundary clause deliberately. ADR-0020's
# missing-boundary-clause SUGGESTION fires on any description without one, so
# a fixture that omits it is never "otherwise clean" — every test asserting
# SUGGESTION-freedom would be asserting the boundary check's absence instead
# of the thing it names. "anything else" is not hyphenated, so the clause adds
# a boundary marker without adding a routing target to resolve.
#
# metadata.version is equally load-bearing: ADR-0022 makes it mandatory and
# validate.sh FAILs without it, so a fixture omitting it would not be
# "otherwise clean" either.
make_valid_skill() {
local dir="$1"
local name
name="$(basename "$dir")"
mkdir -p "$dir/scripts"
cat > "$dir/SKILL.md" <<EOF
---
name: $name
description: A valid skill description that is well within the limit. Do not use for anything else.
metadata:
version: "1.0.0"
---
## Step 1
Do the thing.
EOF
}
# Helper: a description of EXACTLY <n> characters that carries a boundary
# clause and names no routing target. The tests below measure the description
# LENGTH, so the clause has to be paid for out of the same budget rather than
# appended to it — hence the padding arithmetic instead of a fixed suffix.
desc_of_length() {
python3 - "$1" <<'PY'
import sys
n = int(sys.argv[1])
prefix = 'Use when doing the thing. Do not use for anything else. '
assert n >= len(prefix), 'requested description shorter than the boundary clause'
print(prefix + 'x' * (n - len(prefix)))
PY
}
# Helper: create a skill directory with an exact description length and an
# exact body word count. <desc> is used verbatim; <body_words> "word"
# tokens follow the frontmatter. Used by the ADR-0020 boundary tests.
make_sized_skill() {
local dir="$1" desc="$2" body_words="$3"
local name
name="$(basename "$dir")"
mkdir -p "$dir"
{
echo "---"
echo "name: $name"
echo "description: $desc"
echo "metadata:"
echo ' version: "1.0.0"'
echo "---"
echo ""
python3 -c "print(' '.join(['word'] * $body_words))"
} > "$dir/SKILL.md"
}
# Helper: build a self-contained fixture plugin tree so the boundary-target
# resolver has a real authoring source to resolve against, independent of
# this repo's live skills. Echoes the subject skill's directory.
#
# <root>/plugins/fixture-plugin/.apm/skills/<subject>/SKILL.md
# <root>/plugins/fixture-plugin/.apm/skills/fixture-sibling-skill/SKILL.md
# <root>/plugins/fixture-plugin/.apm/agents/fixture-sibling-agent.agent.md
#
# The sibling gets a real SKILL.md, and that is load-bearing rather than
# tidiness: a directory under skills/ is a resolvable name only when it
# HOLDS one. An empty leftover directory is untracked by git, so counting
# one made a target resolve on the machine that made it and dangle in a
# fresh clone. This helper used to mkdir the sibling and write nothing into
# it, so the corroborator every blocking-tier test depends on silently
# stopped resolving the moment that rule was enforced.
make_fixture_tree() {
local root="$1" subject="$2"
local apm="$root/plugins/fixture-plugin/.apm"
mkdir -p "$apm/skills/$subject" "$apm/skills/fixture-sibling-skill" "$apm/agents"
touch "$apm/agents/fixture-sibling-agent.agent.md"
make_sized_skill "$apm/skills/fixture-sibling-skill" \
"Use when doing the other thing. Do not use for anything else." 10
echo "$apm/skills/$subject"
}
}
teardown() {
rm -rf "$TMPDIR"
}
# ---------------------------------------------------------------------------
# Passing cases
# ---------------------------------------------------------------------------
@test "passes on a valid minimal skill" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
run bash "$SCRIPT" "$skill"
assert_success
}
@test "--help exits 0" {
run bash "$SCRIPT" --help
assert_success
assert_output --partial "Usage:"
}
@test "passes when scripts/ directory is absent" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
rmdir "$skill/scripts"
run bash "$SCRIPT" "$skill"
assert_success
}
@test "FILL IN: inside backticks does not fail" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
echo "Use \`FILL IN: value\` as an example." >> "$skill/SKILL.md"
run bash "$SCRIPT" "$skill"
assert_success
}
@test "the 1024-char agentskills.io spec backstop is unchanged and separate from the ADR-0020 ceiling" {
local skill="$TMPDIR/my-skill"
local name
name="$(basename "$skill")"
mkdir -p "$skill"
local desc
desc="$(python3 -c "print('x' * 1024)")"
cat > "$skill/SKILL.md" <<EOF
---
name: $name
description: $desc
---
## Step 1
Do the thing.
EOF
run bash "$SCRIPT" "$skill"
# Two independent gates on one value: the spec limit still PASSES at
# exactly 1024 (its own boundary is unmoved), while ADR-0020's 400-char
# ceiling FAILs. The run fails on the second, not the first.
assert_output --partial "description length 1024 chars (agentskills.io spec limit: 1024)"
assert_output --partial "400-character ADR-0020 ceiling"
assert_failure
}
@test "passes at exactly 500 lines" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
local current
current="$(wc -l < "$skill/SKILL.md")"
local needed=$(( 500 - current ))
python3 -c "print('\n' * $needed, end='')" >> "$skill/SKILL.md"
run bash "$SCRIPT" "$skill"
assert_success
}
# ---------------------------------------------------------------------------
# Failing cases
# ---------------------------------------------------------------------------
@test "fails when SKILL.md is missing" {
local skill="$TMPDIR/my-skill"
mkdir -p "$skill"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when name does not match directory" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
sed -i 's/^name: .*/name: wrong-name/' "$skill/SKILL.md"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when description exceeds 1024 chars" {
local skill="$TMPDIR/my-skill"
local name
name="$(basename "$skill")"
mkdir -p "$skill"
local desc
desc="$(python3 -c "print('x' * 1025)")"
cat > "$skill/SKILL.md" <<EOF
---
name: $name
description: $desc
---
## Step 1
Do the thing.
EOF
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when SKILL.md exceeds 500 lines" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
python3 -c "print('\n' * 500)" >> "$skill/SKILL.md"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when body contains unfilled FILL IN: placeholder" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
echo "FILL IN: replace this" >> "$skill/SKILL.md"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when a script is not executable" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
echo "#!/usr/bin/env bash" > "$skill/scripts/helper.sh"
chmod -x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when a script has an interactive prompt" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
printf '#!/usr/bin/env bash\nread -p "Enter value: " VAL\n' > "$skill/scripts/helper.sh"
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when a script reads a variable with no redirect" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
printf '#!/usr/bin/env bash\nread -r ANSWER\n' > "$skill/scripts/helper.sh"
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when an interactive prompt string contains an angle bracket" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
printf '#!/usr/bin/env bash\nread -p "enter <name>: " NAME\n' > "$skill/scripts/helper.sh"
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "passes when a script reads from a here-string" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
printf '#!/usr/bin/env bash\nLINE="a b"\nread -r X Y <<< "$LINE"\n' \
> "$skill/scripts/helper.sh"
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_success
}
@test "passes when a script reads from a here-doc" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
printf '#!/usr/bin/env bash\nread -r X <<EOF\nvalue\nEOF\n' > "$skill/scripts/helper.sh"
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_success
}
@test "passes when a script reads from a file redirect" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
printf '#!/usr/bin/env bash\nread -r LINE < "$1"\n' > "$skill/scripts/helper.sh"
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_success
}
@test "passes when a script reads from a pipe continued onto the next line" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
printf '#!/usr/bin/env bash\nprintf %%s "$1" |\n read -r X\n' > "$skill/scripts/helper.sh"
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_success
}
@test "prose inside a usage() here-doc that wraps onto a line starting with 'read' does not fail" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
# The exact shape that made skill-audit hard-FAIL on its own
# validate-provenance.sh: a usage() heredoc whose wrapped sentence puts the
# English verb "read" in column 0.
cat > "$skill/scripts/helper.sh" <<'SH'
#!/usr/bin/env bash
usage() {
cat <<EOF
Checks performed:
4 A Contributing files block this parser cannot
read is reported as an INFO, never skipped silently.
EOF
}
usage
SH
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_success
}
@test "a real interactive read AFTER a here-doc is still caught" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
# Pins that the here-doc exemption ends at its terminator. A body skip that
# ran to end-of-file would swallow this read and report the script clean.
cat > "$skill/scripts/helper.sh" <<'SH'
#!/usr/bin/env bash
cat <<EOF
read this text
EOF
read -r ANSWER
SH
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "an unterminated here-doc opener does not disarm the check for the rest of the file" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
# `<<` here is inside a string, not an opener. Treating it as one would skip
# every following line — a false negative, the direction this check must
# never fail in.
cat > "$skill/scripts/helper.sh" <<'SH'
#!/usr/bin/env bash
echo "shift left with a << b"
read -r ANSWER
SH
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "a bare input() inside an embedded-python here-doc is still caught" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
# The here-doc exemption is for the `read` heuristic only: these scripts
# embed Python in a here-doc as a matter of course, so exempting the body
# wholesale would disarm the check across the corpus.
cat > "$skill/scripts/helper.sh" <<'SH'
#!/usr/bin/env bash
python3 - <<'PY'
print("press enter")
input()
PY
SH
chmod +x "$skill/scripts/helper.sh"
run bash "$SCRIPT" "$skill"
assert_failure
}
# ---------------------------------------------------------------------------
# ADR-0022 — metadata.version is mandatory. FAIL tier, matching the
# skill-size-check pre-commit hook: an audit that graded this lower would
# report ready-to-ship on a file the commit gate rejects.
# ---------------------------------------------------------------------------
@test "ADR-0022: a SKILL.md with no metadata block at all FAILs" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
python3 - "$skill/SKILL.md" <<'PY'
import sys
p = sys.argv[1]
s = open(p).read().replace('metadata:\n version: "1.0.0"\n', '')
open(p, 'w').write(s)
PY
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "metadata.version"
}
@test "ADR-0022: a metadata block with no version key FAILs" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
python3 - "$skill/SKILL.md" <<'PY'
import sys
p = sys.argv[1]
s = open(p).read().replace(' version: "1.0.0"\n', ' category: factory\n')
open(p, 'w').write(s)
PY
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "metadata.version"
}
@test "ADR-0022: a two-part metadata.version FAILs as malformed, not passes as present" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
python3 - "$skill/SKILL.md" <<'PY'
import sys
p = sys.argv[1]
s = open(p).read().replace(' version: "1.0.0"\n', ' version: 1.0\n')
open(p, 'w').write(s)
PY
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "three-part semver"
}
@test "ADR-0022: an unquoted three-part metadata.version passes" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
python3 - "$skill/SKILL.md" <<'PY'
import sys
p = sys.argv[1]
s = open(p).read().replace(' version: "1.0.0"\n', ' version: 0.1.3\n')
open(p, 'w').write(s)
PY
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "metadata.version present: '0.1.3'"
}
@test "ADR-0022: a leading zero in the patch part FAILs (1.0.08)" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
python3 - "$skill/SKILL.md" <<'PY'
import sys
p = sys.argv[1]
s = open(p).read().replace(' version: "1.0.0"\n', ' version: "1.0.08"\n')
open(p, 'w').write(s)
PY
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "three-part semver"
}
@test "ADR-0022: a leading zero in the major part FAILs (01.0.1)" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
python3 - "$skill/SKILL.md" <<'PY'
import sys
p = sys.argv[1]
s = open(p).read().replace(' version: "1.0.0"\n', ' version: "01.0.1"\n')
open(p, 'w').write(s)
PY
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "three-part semver"
}
@test "ADR-0022: a multi-digit part with no leading zero passes (1.0.10)" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
python3 - "$skill/SKILL.md" <<'PY'
import sys
p = sys.argv[1]
s = open(p).read().replace(' version: "1.0.0"\n', ' version: "1.0.10"\n')
open(p, 'w').write(s)
PY
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "metadata.version present: '1.0.10'"
}
@test "ADR-0022: a zero major part passes (0.1.0)" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
python3 - "$skill/SKILL.md" <<'PY'
import sys
p = sys.argv[1]
s = open(p).read().replace(' version: "1.0.0"\n', ' version: "0.1.0"\n')
open(p, 'w').write(s)
PY
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "metadata.version present: '0.1.0'"
}
@test "fails when name contains consecutive hyphens" {
local skill="$TMPDIR/my--skill"
make_valid_skill "$skill"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when name has a leading hyphen" {
local skill="$TMPDIR/-my-skill"
make_valid_skill "$skill"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when no frontmatter block is present" {
local skill="$TMPDIR/my-skill"
mkdir -p "$skill"
echo "Just some content with no frontmatter." > "$skill/SKILL.md"
run bash "$SCRIPT" "$skill"
assert_failure
}
@test "fails when no arguments are given" {
run bash "$SCRIPT"
assert_failure
}
# ---------------------------------------------------------------------------
# ADR-0020 — description budget (250 SUGGESTION / 400 FAIL)
#
# These sit UNDER the agentskills.io 1024-character spec backstop above, which
# is unchanged. Both ceilings are inclusive: exactly at the number passes that
# tier, one past it trips.
# ---------------------------------------------------------------------------
@test "ADR-0020: description of exactly 250 chars raises no suggestion" {
local skill="$TMPDIR/my-skill"
make_sized_skill "$skill" "$(desc_of_length 250)" 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "SUGGESTION"
}
@test "ADR-0020: description of 251 chars raises a SUGGESTION and still exits 0" {
local skill="$TMPDIR/my-skill"
make_sized_skill "$skill" "$(desc_of_length 251)" 10
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "SUGGESTION"
assert_output --partial "description is 251 chars"
assert_output --partial "All checks passed (1 suggestion(s))."
}
@test "ADR-0020: description of exactly 400 chars is a SUGGESTION, not a FAIL" {
local skill="$TMPDIR/my-skill"
make_sized_skill "$skill" "$(desc_of_length 400)" 10
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "SUGGESTION"
}
@test "ADR-0020: description of 401 chars FAILs and exits non-zero" {
local skill="$TMPDIR/my-skill"
make_sized_skill "$skill" "$(desc_of_length 401)" 10
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "description is 401 chars"
assert_output --partial "400-character ADR-0020 ceiling"
}
@test "ADR-0020: description length is measured after YAML folding is resolved" {
local skill="$TMPDIR/my-skill"
mkdir -p "$skill"
# A >-folded block scalar: 11 lines of 40 chars folded with 10 joining
# spaces = 450 characters. Measured off its raw `description: >` line it is
# 1 character and passes; measured as the folded VALUE it must FAIL. This
# is exactly the case a line-wise regex gets wrong.
{
echo "---"
echo "name: my-skill"
echo "description: >"
python3 -c "print('\n'.join([' ' + 'x' * 40] * 11))"
echo "---"
echo ""
echo "Do the thing."
} > "$skill/SKILL.md"
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "description is 450 chars"
assert_output --partial "400-character ADR-0020 ceiling"
}
# ---------------------------------------------------------------------------
# ADR-0020 — body budget (600 SUGGESTION / 900 FAIL), body ONLY
#
# Distinct from the 2,770-word whole-file spec ceiling above, which counts
# frontmatter too and is unchanged. Do not unify them.
# ---------------------------------------------------------------------------
@test "ADR-0020: body of exactly 600 words raises no suggestion" {
local skill="$TMPDIR/my-skill"
make_sized_skill "$skill" "A short valid description. Do not use for anything else." 600
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "SUGGESTION"
}
@test "ADR-0020: body of 601 words raises a SUGGESTION and still exits 0" {
local skill="$TMPDIR/my-skill"
make_sized_skill "$skill" "A short valid description. Do not use for anything else." 601
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "body is 601 words"
assert_output --partial "All checks passed (1 suggestion(s))."
}
@test "ADR-0020: body of exactly 900 words is a SUGGESTION, not a FAIL" {
local skill="$TMPDIR/my-skill"
make_sized_skill "$skill" "A short valid description. Do not use for anything else." 900
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "body is 900 words"
}
@test "ADR-0020: body of 901 words FAILs and exits non-zero" {
local skill="$TMPDIR/my-skill"
make_sized_skill "$skill" "A short valid description. Do not use for anything else." 901
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "body is 901 words"
assert_output --partial "900-word ADR-0020 ceiling"
}
@test "ADR-0020: the body gate counts the body only — frontmatter words do not count toward it" {
local skill="$TMPDIR/my-skill"
# 895 body words plus a description long enough that the WHOLE FILE is well
# over 900 words. The body gate must stay silent; the 2,770-word whole-file
# ceiling is a separate measurement and is nowhere near tripping.
make_sized_skill "$skill" "$(python3 -c "print(' '.join(['w'] * 100))")" 895
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "900-word ADR-0020 ceiling"
}
# ---------------------------------------------------------------------------
# ADR-0020 — resolvable boundary targets
#
# Resolved against the AUTHORING SOURCE (plugins/*/.apm/skills/ and
# plugins/*/.apm/agents/), never .claude/skills/, so the check works offline and
# before an apm install. Every fixture below builds its own plugin tree rather
# than leaning on this repo's live skills.
# ---------------------------------------------------------------------------
@test "ADR-0020: a boundary target naming an existing sibling skill resolves" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use fixture-sibling-skill instead." 10
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "boundary target(s) resolve"
}
@test "ADR-0020: a boundary target naming a non-existent skill FAILs when its sentence names one that resolves" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
# `fixture-sibling-skill` is the corroborator: a prose-form target only earns
# a FAIL when its own sentence proves it is a routing sentence. See the
# shared resolver's CORROBORATION note, and the uncorroborated case below.
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use fixture-sibling-skill or fixture-missing-skill instead." 10
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "routes to 'fixture-missing-skill'"
}
@test "ADR-0020: a LONE boundary target naming a non-existent skill is a SUGGESTION, not a FAIL" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
# Same grammar as the case above and as "run \`pre-commit\` instead" — a
# route verb, a hyphenated name, terminal position. Nothing local separates a
# broken route from a tool name, so the target is named on every run but does
# not block: this gate ships with no baseline and no suppression mechanism.
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use fixture-missing-skill instead." 10
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "SUGGESTION"
assert_output --partial "routes to 'fixture-missing-skill'"
refute_output --partial "FAIL description routes to"
}
@test "ADR-0020: a boundary target naming an AGENT file resolves (agents are valid routing targets)" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
make_sized_skill "$skill" "Use when doing the thing. Do not use when the caller is an agent — invoke fixture-sibling-agent instead." 10
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "boundary target(s) resolve"
}
@test "ADR-0020: a /slash-command boundary target that does not resolve FAILs" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
make_sized_skill "$skill" "Use when doing the thing. Do not use when improvements are wanted — use /fixture-missing-improve instead." 10
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "routes to 'fixture-missing-improve'"
}
@test "ADR-0020: a backticked name that does not resolve FAILs when its sentence names one that resolves" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
make_sized_skill "$skill" "Use when doing the thing. Composes \`fixture-sibling-skill\` and \`fixture-missing-helper\` for the shared part." 10
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "routes to 'fixture-missing-helper'"
}
@test "ADR-0020: a /slash-command target is route NOTATION and FAILs on its own, uncorroborated" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
# The escape hatch from the SUGGESTION tier: `/name` and `-> name` are never
# how prose cites a tool, so they are exempt from corroboration. An author
# who wants a route checked unconditionally writes one of those two forms.
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use /fixture-missing-notation instead." 10
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "routes to 'fixture-missing-notation'"
}
@test "ADR-0020: a bare hyphenated word outside a boundary sentence is not read as a routing target" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
# "run pre-commit hooks" is pc-run's real phrasing. A naive extractor reads
# it as a route to a non-existent `pre-commit` skill.
make_sized_skill "$skill" "Use when the user wants to run pre-commit hooks or install git hooks." 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "pre-commit"
}
@test "ADR-0020: an arrow chain outside a boundary clause is not read as a routing target" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
# diagnose's real process chain. Only ADR-0020's `Not <thing> -> <skill>`
# form makes a bare arrow target a route.
make_sized_skill "$skill" "Reproduce → minimise → instrument → fix → regression-test. Use when a bug is reported." 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "regression-test"
}
@test "ADR-0020: MCP tool names and capitalised tool names are not read as routing targets" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
make_sized_skill "$skill" "Use when writing issues. Do not use for local files (use Read/Write/Edit) — that write goes through \`issue_write\`/\`pull_request_write\` instead." 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "routes to"
}
@test "ADR-0020: ADR's compressed boundary form (Not <thing> -> <skill>) is checked" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
make_sized_skill "$skill" "Use when doing the thing. Not the other thing → fixture-missing-target." 10
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "routes to 'fixture-missing-target'"
}
@test "ADR-0020: the boundary check declines rather than false-FAILs when no authoring source is found" {
# Deliberately NOT built with make_fixture_tree: this skill sits in a bare
# temp directory with no plugins/*/.apm/ above it and no .git, so the resolver
# legitimately has no universe. That is a real path (a skill being drafted
# outside any repo), and the required behaviour is to DECLINE OUT LOUD rather
# than either false-FAIL or pass in silence — silence is what let a whole gate
# family go missing unnoticed. So the INFO text and the named unchecked target
# are both asserted, not just the absence of a failure.
local skill="$TMPDIR/orphan/my-skill"
make_sized_skill "$skill" "Use when doing the thing. Do not use for the other thing — use some-other-skill instead." 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "routes to"
assert_output --partial "boundary-target resolution DID NOT RUN"
assert_output --partial "Unchecked target(s): some-other-skill"
}
# ---------------------------------------------------------------------------
# ADR-0020 — the hand-invocation carve-out (issue #108)
#
# A skill carrying `disable-model-invocation: true` is absent from the
# model-visible listing entirely: not preloaded, and the Skill tool refuses to
# call it. Its description is never matched against user intent, so
# references/skill-description-quality.md Step 0 gives it ONE plain human-facing
# sentence — no trigger list, no boundary clause — and calls a
# missing-boundary-clause finding on such a skill "a wrong finding, not a strict
# one". Until this ran, nothing here knew the field existed, so the audit
# reported exactly the shape its own rubric mandates, with advice naming a
# router that cannot see the skill.
#
# The carve-out is narrow. Both size gates are unaffected and both are pinned
# below: the body is loaded on invocation like any other body, and the
# 400-character ceiling is an outlier stop rather than a routing budget.
# ---------------------------------------------------------------------------
# Helper: a skill directory carrying `disable-model-invocation: true`.
make_hand_invoked_skill() {
local dir="$1" desc="$2" body_words="$3"
local name
name="$(basename "$dir")"
mkdir -p "$dir"
{
echo "---"
echo "name: $name"
echo "description: $desc"
echo "disable-model-invocation: true"
echo "metadata:"
echo ' version: "1.0.0"'
echo "---"
echo ""
python3 -c "print(' '.join(['word'] * $body_words))"
} > "$dir/SKILL.md"
}
@test "ADR-0020: a hand-invoked skill is not asked for a boundary clause" {
local skill="$TMPDIR/my-skill"
make_hand_invoked_skill "$skill" \
"Tell the agent to zoom out and give broader context or a higher level perspective." 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "has no boundary clause"
assert_output --partial "hand-invoked"
}
@test "ADR-0020: the SAME description without the flag IS asked for a boundary clause" {
# The control. Without it the case above is satisfied by an audit that
# stopped checking boundary clauses altogether.
local skill="$TMPDIR/my-skill"
make_sized_skill "$skill" \
"Tell the agent to zoom out and give broader context or a higher level perspective." 10
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "has no boundary clause"
}
@test "ADR-0020: a hand-invoked skill is exempt from the 250-character description target" {
local skill="$TMPDIR/my-skill"
make_hand_invoked_skill "$skill" \
"$(python3 -c "print('Tell the agent to zoom out. ' + 'x' * 273)")" 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "over the 250-character"
}
@test "ADR-0020: a hand-invoked description over 400 chars still FAILS" {
# The half the carve-out does NOT lift. 400 is an outlier stop, not a
# routing-quality target: a hand-invoked description is still the one line
# the user reads when choosing from the `/` menu.
local skill="$TMPDIR/my-skill"
make_hand_invoked_skill "$skill" \
"$(python3 -c "print('Tell the agent to zoom out. ' + 'x' * 374)")" 10
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "400-character"
}
@test "ADR-0020: a hand-invoked body over 900 words still FAILS" {
# The body is loaded on invocation exactly like any other body and competes
# with the caller's live conversation the same way, so no body tier moves.
local skill="$TMPDIR/my-skill"
make_hand_invoked_skill "$skill" "Tell the agent to zoom out." 901
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "900-word"
}
# ---------------------------------------------------------------------------
# ADR-0020 — one arrow, one target (issue #107)
#
# Only the FIRST target after an arrow was resolved: the conjunction
# continuation is wired to the prose route verbs and never to arrows. So this
# script printed "1 of 1 boundary target(s) resolve" on a clause naming two,
# and the second was resolved by nothing and reported by nothing. A typo in it
# shipped through a green gate. The shape is now rejected rather than the
# extractor widened.
# ---------------------------------------------------------------------------
@test "ADR-0020: an arrow clause naming two targets is reported, not silently half-checked" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
# A bare `Not ... ->` sentence carries no BOUNDARY_MARKER, so the backtick
# sweep does not run and the second target is invisible to every other rule
# in the resolver — this is the exact shape #107 measured.
make_sized_skill "$skill" "Use when doing the thing. Not the other thing -> \`fixture-sibling-skill\` or \`fixture-missing-second\`." 10
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "names more than one target"
}
@test "ADR-0020: one arrow per target — the convention the suggestion asks for — is silent" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
make_sized_skill "$skill" "Use when doing the thing. Not the other thing -> \`fixture-sibling-skill\`." 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "names more than one target"
}
# ---------------------------------------------------------------------------
# ADR-0020 — a dotted filename in a boundary clause (issue #110)
#
# `[^.;]` could not cross the `.` in `AGENTS.md`, so a clause naming a dotted
# file between "Not" and the arrow was invisible. With a backticked target that
# was a MISDIAGNOSIS — "no boundary clause" reported on a clause that was
# present and working. With a BARE target it was worse: the target was never
# extracted, so the dangling check silently did not run on it.
# ---------------------------------------------------------------------------
@test "ADR-0020: a boundary clause naming a dotted filename is not reported as missing" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
make_sized_skill "$skill" "Use when doing the thing. Not AGENTS.md -> \`fixture-sibling-skill\`." 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "has no boundary clause"
assert_output --partial "description has a boundary clause"
}
@test "ADR-0020: a BARE target after a dotted filename is extracted and checked" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
# The silent half of #110: this clause produced no target at all, so it was
# neither resolved nor reported — a route to a non-existent skill shipping
# through a green gate with no finding of any kind.
make_sized_skill "$skill" "Use when doing the thing. Not AGENTS.md -> fixture-missing-dotted." 10
run bash "$SCRIPT" "$skill"
assert_failure
assert_output --partial "routes to 'fixture-missing-dotted'"
}
@test "ADR-0020: an arrow clause yielding no target is reported as unparsed, not as missing" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
# A single-word target is deliberately not matchable bare, because
# `research`, `triage` and `forge` are all skill names AND ordinary English.
# The clause is present; saying it is missing sends the author to add a
# second copy of a clause that is already there.
make_sized_skill "$skill" "Use when doing the thing. Not the other thing -> forge." 10
run bash "$SCRIPT" "$skill"
assert_success
refute_output --partial "has no boundary clause"
assert_output --partial "no target could be read"
}
# ---------------------------------------------------------------------------
# ADR-0020 — the unparsed diagnostic is PER CLAUSE, not per description
#
# boundary_clause_status() used to test `BOUNDARY_ARROW.search(description) and
# not _arrow_targets(description)`. Both operands took the WHOLE description,
# so ONE arrow clause that parsed suppressed the diagnostic for every other
# clause beside it.
#
# The shape that hides there is a backticked hyphenated target wrapped across
# the line break of a `>` folded scalar: the fold turns `` `fixture-sibling- ``
# / `` skill` `` into `fixture-sibling- skill`, which no extractor can read.
# Written BARE the same wrap is reported correctly, so the two spellings
# disagreed. 26 of this repo's 38 skill descriptions carry more than one arrow
# clause, which is how wide the suppression was. This is the #100 regression
# class: no ERROR, no SUGGESTION, exit 0.
# ---------------------------------------------------------------------------
@test "ADR-0020: an unparsed arrow clause is reported even when a sibling clause parses" {
local skill
skill="$(make_fixture_tree "$TMPDIR/tree" "my-skill")"
# Written by hand rather than through make_sized_skill: the `>` folded
# scalar and the wrap INSIDE the backticks are the fixture. The second
# clause parses and resolves against the fixture sibling, and that is what
# used to silence the first.
cat > "$skill/SKILL.md" <<'EOF'
---
name: my-skill
description: >
Use when doing the thing. Not the other thing -> `fixture-sibling-
skill`. Not a third thing -> `fixture-sibling-skill`.
metadata:
version: "1.0.0"
---
word word word word word word word word word word
EOF
run bash "$SCRIPT" "$skill"
assert_success
assert_output --partial "no target could be read"
refute_output --partial "has no boundary clause"
}
# ---------------------------------------------------------------------------
# Encoding, write side: sys.stdout/stderr.reconfigure(encoding='utf-8')
#
# read_text() in the shared resolver block pins the READS to UTF-8. That moved
# the LC_ALL=C crash to the WRITE: this script's own message text carries em
# dashes (the ADR-0020 boundary SUGGESTION is one), so the streams' ASCII
# default raised UnicodeEncodeError while PRINTING — after every check had
# already run, losing the whole report at the last step.
# ---------------------------------------------------------------------------
@test "under LC_ALL=C the report is printed, not lost to a UnicodeEncodeError" {
local dir="$TMPDIR/locale-skill"
mkdir -p "$dir"
cat > "$dir/SKILL.md" <<EOF
---
name: locale-skill
description: A valid skill description that is well within the limit.
metadata:
version: "1.0.0"
---
## Step 1
Do the thing.
EOF
run env LC_ALL=C PYTHONUTF8=0 bash "$SCRIPT" "$dir"
assert_success
assert_output --partial "description has no boundary clause"
refute_output --partial "UnicodeEncodeError"
refute_output --partial "Traceback"
}
# ---------------------------------------------------------------------------
# Auto-detection — the merged entry point classifies its own target
#
# NEW with the factory-audit merge, and new behaviour rather than a ported
# case: scripts/validate.sh is now ONE entry point for both artifact types and
# works out from the target which rubric to run. A DIRECTORY holding SKILL.md is
# a skill; a FILE named *.agent.md, or sitting under .apm/agents/, is an agent.
#
# Before the merge each script was hard-wired to one type, so there was nothing
# here that could be wrong. Now a misclassification is silent and total: the
# wrong rubric runs end to end and reports the artifact clean against gates that
# never applied to it, while every gate that did apply goes unrun. Nothing else
# in this suite would notice, because every other fixture is a skill directory
# and would be classified correctly even by a detector that always guessed
# "skill".
#
# The agent half of the same contract is pinned from the other side, in
# validate-agent.bats — same script, same detector, agent-shaped fixtures.
# ---------------------------------------------------------------------------
@test "auto-detect: a directory holding SKILL.md is audited in SKILL mode" {
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
run bash "$SCRIPT" "$skill"
assert_success
# ADR-0022's metadata.version is mandatory for skills and has no agent
# analogue whatsoever, so this line is positive proof the SKILL rubric ran —
# not merely that the run survived. A bare assert_success would be satisfied
# by a detector that classified this as an agent and found nothing to say.
assert_output --partial "metadata.version present"
# 'counterpart' is agent-mode vocabulary (the CC/Copilot pair check). A skill
# directory must never reach a check that has a concept of a counterpart.
refute_output --partial "counterpart"
}
@test "auto-detect: a SKILL.md FILE path is audited in SKILL mode, not rejected" {
# NEW with the merge and additive rather than ported: pre-merge, handing the
# SKILL.md itself to skill-audit's validate.sh hit the directory precondition
# and gave a useless exit 1. It matters because pre-commit `files:` hooks
# match FILES — the vale-audit-prefilter-skill hook's regex ends in
# /SKILL\.md$ — so every hook-driven invocation hands over a SKILL.md
# path, never the directory above it. The entry point resolves the file to
# its directory before dispatching.
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
run bash "$SCRIPT" "$skill/SKILL.md"
assert_success
# Same positive proof as the directory case: metadata.version is skill-only
# and has no agent analogue, so this line says the SKILL rubric ran.
assert_output --partial "metadata.version present"
refute_output --partial "counterpart"
}
@test "auto-detect: a SKILL.md FILE path and its directory produce the same verdict" {
# The resolution must be transparent, not merely non-fatal. If the two
# spellings of one target could disagree, a pre-commit run and a hand run
# would report differently on the same skill and neither would look wrong.
# The name check is the sharp end: it compares `name` against the DIRECTORY
# basename, so a target left as the file would compare against "SKILL.md".
local skill="$TMPDIR/my-skill"
make_valid_skill "$skill"
run bash "$SCRIPT" "$skill"
local dir_status="$status"
local dir_output="$output"
run bash "$SCRIPT" "$skill/SKILL.md"
[ "$status" -eq "$dir_status" ]
[ "$output" = "$dir_output" ]
assert_output --partial "name 'my-skill' matches directory 'my-skill'"
}
@test "auto-detect: a target that is neither a skill directory nor an agent file FAILs, naming what it was handed" {
# The detector must not guess. Falling back to either rubric on an
# unclassifiable target yields a verdict about rules that were never meant
# to apply, and exiting 0 publishes that verdict as a pass — the worst of
# the three possible outcomes, because it is the silent one.
#
# A plain .txt file is neither shape under any reading of the contract: not
# a directory holding SKILL.md, not *.agent.md, not under .apm/agents/. The
# directory flavour of the same mismatch — a directory with no SKILL.md — is
# pinned separately by "fails when SKILL.md is missing" above.
local dir="$TMPDIR/neither"
mkdir -p "$dir"
echo "not an artifact of either kind" > "$dir/notes.txt"
run bash "$SCRIPT" "$dir/notes.txt"
assert_failure
# Non-zero is necessary but not sufficient: a non-zero exit with nothing on
# stdout is indistinguishable from a clean-but-failing run, which is the
# confusion the exit-2 tier in the provenance suites was created to end.
refute_output ""
assert_output --partial "$dir/notes.txt"
}
@test "entry point: a missing resolver library in skill mode exits 2 naming it, never exit 1" {
# validate-agent.bats pins this for agent mode; each mode sources its own
# libraries, so each mode's guard is pinned separately.
make_valid_skill "$TMPDIR/my-skill"
local lone="$TMPDIR/lone-scripts"
cp -R "$(dirname "$SCRIPT")" "$lone"
rm "$lone/lib-boundary-resolver.sh"
run bash "$lone/validate.sh" "$TMPDIR/my-skill"
[ "$status" -eq 2 ]
assert_output --partial "required library '$lone/lib-boundary-resolver.sh' is missing or unreadable"
refute_output --partial "No such file or directory"
}