Files
holocron/tests/test-adr0020-contract.sh
Defame1297 27a76692b0 fix(kyberforge): stop the provenance checker skipping check 8 on unparsed input
484357a taught the Contributing parser the bullet form, but a block it still
could not parse returned the same empty result as an explicit "(none)", so the
checker read "no contributing files" and skipped check 8 rather than reporting
that it could not tell. Checks 7 and 8 were consequently dead across the whole
git plugin without anything failing.

The parser now distinguishes "declared none" from "could not parse", which
wakes both checks. Because the parser is duplicated between the skill-audit and
agent-audit copies, it is fenced with BEGIN/END markers and a test hashes the
two regions so the copies cannot drift apart again silently.

Addresses #111.
2026-08-31 08:01:18 +00:00

398 lines
18 KiB
Bash
Executable File

#!/usr/bin/env bash
# Regression test for the STRUCTURAL claims the ADR-0020 gate family makes about
# itself. None of them was pinned anywhere before this file, and each one fails
# silently — which is the whole reason they need a test rather than a comment:
#
# 1. "ONE resolver, embedded VERBATIM in three scripts." The block between the
# BEGIN/END markers is copied, not imported, because a cache-installed
# plugin's scripts cannot read files outside their own plugin directory.
# Nothing but this file asserts the three copies are still identical, and a
# one-line edit to a single copy is invisible: every constant-agreement
# assertion in tests/test-skill-size-check.sh still passes, because the
# CONSTANTS are not what drifted.
# 1b. The same claim, one directory over, for the Contributing-files parser
# embedded in both validate-provenance.sh copies. That one was worse: the
# agent-audit copy's docstring ASSERTED it was kept behaviourally identical
# to skill-audit's, and the two had already drifted.
# 2. Both interpreter preflights, in all three scripts. python3 and PyYAML are
# declared HARD dependencies precisely so a missing one cannot turn into a
# vacuous pass, and the two are checked separately so the message names the
# thing to install rather than the wrong one.
# 3. `verbose: true` on the skill-size-check hook. It is the ENTIRE delivery
# mechanism for the SUGGESTION tier: pre-commit prints nothing at all for a
# passing hook, and a SUGGESTION deliberately does not fail, so dropping
# one word from the config silences the tier ADR-0020 depends on while
# every test and every hook still reports green.
set -euo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
HOOK="$REPO_ROOT/scripts/skill-size-check.sh"
SKILL_VALIDATE="$REPO_ROOT/plugins/kyberforge/.apm/skills/skill-audit/scripts/validate.sh"
AGENT_VALIDATE="$REPO_ROOT/plugins/kyberforge/.apm/skills/agent-audit/scripts/validate.sh"
PASS=0
FAIL=0
pass() { echo " PASS: $1"; PASS=$((PASS + 1)); }
fail() { echo " FAIL: $1"; FAIL=$((FAIL + 1)); }
TMPDIR_T="$(mktemp -d)"
trap 'rm -rf "$TMPDIR_T"' EXIT
BEGIN_MARKER='# ===== BEGIN ADR-0020 SHARED BOUNDARY RESOLVER ====='
END_MARKER='# ===== END ADR-0020 SHARED BOUNDARY RESOLVER ====='
# ---------------------------------------------------------------------------
# 1. The shared resolver block is byte-identical in all three scripts
# ---------------------------------------------------------------------------
echo ""
echo "--- the ADR-0020 shared resolver block is byte-identical in all three scripts ---"
# Marker discipline first. An unbalanced or duplicated marker pair makes the
# extraction below silently measure the wrong span — a sed range that never
# closes swallows the rest of the file, and one that opens twice concatenates
# two spans. Both would still compare "equal" if all three were mangled the
# same way, so the shape is asserted before the contents.
MARKERS_OK=true
for f in "$HOOK" "$SKILL_VALIDATE" "$AGENT_VALIDATE"; do
if [[ ! -f "$f" ]]; then
fail "script not found: $f"
MARKERS_OK=false
continue
fi
b="$(grep -cFx "$BEGIN_MARKER" "$f" || true)"
e="$(grep -cFx "$END_MARKER" "$f" || true)"
if [[ "$b" == "1" && "$e" == "1" ]]; then
pass "${f#"$REPO_ROOT/"} carries exactly one BEGIN and one END marker"
else
fail "${f#"$REPO_ROOT/"} has $b BEGIN and $e END markers, expected 1 and 1"
MARKERS_OK=false
fi
done
if ! $MARKERS_OK; then
fail "skipping the byte-identity comparison — the marker pairs are not well-formed, so any extraction would measure the wrong span"
else
HASHES=()
LINECOUNTS=()
for f in "$HOOK" "$SKILL_VALIDATE" "$AGENT_VALIDATE"; do
out="$TMPDIR_T/block-$(echo "$f" | md5sum | cut -c1-8).txt"
sed -n "/^${BEGIN_MARKER}\$/,/^${END_MARKER}\$/p" "$f" > "$out"
HASHES+=("$(md5sum < "$out" | cut -d' ' -f1)")
LINECOUNTS+=("$(wc -l < "$out" | tr -d ' ')")
done
if [[ "${HASHES[0]}" == "${HASHES[1]}" && "${HASHES[1]}" == "${HASHES[2]}" ]]; then
pass "all three copies hash to ${HASHES[0]} (${LINECOUNTS[0]} lines) — agreement by construction, not by coincidence"
else
fail "the shared resolver has DRIFTED: skill-size-check=${HASHES[0]} (${LINECOUNTS[0]} lines), skill-audit=${HASHES[1]} (${LINECOUNTS[1]} lines), agent-audit=${HASHES[2]} (${LINECOUNTS[2]} lines). Edit one copy, then paste it over the other two."
fi
# A block that has been emptied out would hash equal in all three and pass the
# comparison above while enforcing nothing. The resolver is ~570 lines; 100 is
# a floor low enough never to need maintenance and high enough that a gutted
# block cannot sneak past.
if [[ "${LINECOUNTS[0]}" -gt 100 ]]; then
pass "the extracted block is ${LINECOUNTS[0]} lines — the comparison is over real content, not an empty span"
else
fail "the extracted shared block is only ${LINECOUNTS[0]} lines — three identical empty spans would compare equal and assert nothing"
fi
fi
# ---------------------------------------------------------------------------
# 1b. The shared Contributing-files parser is byte-identical in both copies
# ---------------------------------------------------------------------------
# Same defect class, one directory over. parse_contributing_files() is embedded
# in both validate-provenance.sh copies for the same reason the resolver is
# embedded three times, and until this assertion existed the agent-audit copy's
# docstring merely CLAIMED it was "kept behaviourally identical to skill-audit's
# copy" — an invariant nothing checked, and the two had already drifted into
# different spellings of the bullet loop. The parser decides whether checks 4,
# 5 and 8 run at all, so a one-sided edit disables a check in one script while
# every other test stays green.
echo ""
echo "--- the shared Contributing-files parser is byte-identical in both validate-provenance.sh copies ---"
CF_BEGIN='# ===== BEGIN SHARED CONTRIBUTING-FILES PARSER ====='
CF_END='# ===== END SHARED CONTRIBUTING-FILES PARSER ====='
SKILL_PROV="$REPO_ROOT/plugins/kyberforge/.apm/skills/skill-audit/scripts/validate-provenance.sh"
AGENT_PROV="$REPO_ROOT/plugins/kyberforge/.apm/skills/agent-audit/scripts/validate-provenance.sh"
CF_MARKERS_OK=true
for f in "$SKILL_PROV" "$AGENT_PROV"; do
if [[ ! -f "$f" ]]; then
fail "script not found: $f"
CF_MARKERS_OK=false
continue
fi
b="$(grep -cFx "$CF_BEGIN" "$f" || true)"
e="$(grep -cFx "$CF_END" "$f" || true)"
if [[ "$b" == "1" && "$e" == "1" ]]; then
pass "${f#"$REPO_ROOT/"} carries exactly one BEGIN and one END parser marker"
else
fail "${f#"$REPO_ROOT/"} has $b BEGIN and $e END parser markers, expected 1 and 1"
CF_MARKERS_OK=false
fi
done
if ! $CF_MARKERS_OK; then
fail "skipping the parser byte-identity comparison — the marker pairs are not well-formed, so any extraction would measure the wrong span"
else
CF_HASHES=()
CF_LINECOUNTS=()
for f in "$SKILL_PROV" "$AGENT_PROV"; do
out="$TMPDIR_T/cfblock-$(echo "$f" | md5sum | cut -c1-8).txt"
sed -n "/^${CF_BEGIN}\$/,/^${CF_END}\$/p" "$f" > "$out"
CF_HASHES+=("$(md5sum < "$out" | cut -d' ' -f1)")
CF_LINECOUNTS+=("$(wc -l < "$out" | tr -d ' ')")
done
if [[ "${CF_HASHES[0]}" == "${CF_HASHES[1]}" ]]; then
pass "both copies hash to ${CF_HASHES[0]} (${CF_LINECOUNTS[0]} lines) — agreement by construction, not by coincidence"
else
fail "the shared Contributing-files parser has DRIFTED: skill-audit=${CF_HASHES[0]} (${CF_LINECOUNTS[0]} lines), agent-audit=${CF_HASHES[1]} (${CF_LINECOUNTS[1]} lines). Edit one copy, then paste it over the other."
fi
# Two identical EMPTY spans would hash equal and assert nothing, exactly as
# for the resolver above. The parser block is ~93 lines; 40 is a floor low
# enough never to need maintenance and high enough that a gutted block — or
# one reduced to its docstring — cannot sneak past.
if [[ "${CF_LINECOUNTS[0]}" -gt 40 ]]; then
pass "the extracted parser block is ${CF_LINECOUNTS[0]} lines — the comparison is over real content, not an empty span"
else
fail "the extracted parser block is only ${CF_LINECOUNTS[0]} lines — two identical empty spans would compare equal and assert nothing"
fi
fi
# ---------------------------------------------------------------------------
# 2. Both interpreter preflights, in all three scripts
# ---------------------------------------------------------------------------
# The two are checked separately on purpose: `python3 -c 'import yaml'` fails
# identically whether python3 is missing or PyYAML is, and naming the wrong one
# sends the reader to install the wrong thing.
REAL_PYTHON="$(command -v python3)"
# Absolute path, deliberately. The no-python3 fixture below replaces PATH
# wholesale, so a bare `bash` (or `/usr/bin/env bash`) would be resolved against
# that stripped PATH and die with "No such file or directory" before the script
# under test ever starts -- a 127 that looks like the preflight firing.
BASH_BIN="$(command -v bash)"
# A PATH that genuinely has no python3 on it. Built by symlinking the handful of
# binaries the three scripts touch before their own preflight rather than by
# hiding python3 from a full PATH, because there is no portable way to subtract
# one entry from a directory. `bash` is invoked by absolute path below so the
# interpreter itself does not have to be on this PATH.
NOPY_BIN="$TMPDIR_T/nopython-bin"
mkdir -p "$NOPY_BIN"
for b in awk cat cut dirname basename grep sed pwd rm mkdir tr; do
src="$(command -v "$b" 2>/dev/null || true)"
[[ -n "$src" ]] && ln -sf "$src" "$NOPY_BIN/$b"
done
# A python3 that runs but cannot import yaml. A shim on PATH re-execs the real
# interpreter with a PYTHONPATH entry holding a `yaml` module that raises on
# import; PYTHONPATH precedes site-packages on sys.path, so it shadows a real
# PyYAML install without touching it.
SHADOW="$TMPDIR_T/shadow"
mkdir -p "$SHADOW"
printf 'raise ImportError("PyYAML deliberately unavailable in this fixture")\n' \
> "$SHADOW/yaml.py"
NOYAML_BIN="$TMPDIR_T/noyaml-bin"
mkdir -p "$NOYAML_BIN"
cat > "$NOYAML_BIN/python3" <<EOF
#!/bin/sh
PYTHONPATH="$SHADOW\${PYTHONPATH:+:\$PYTHONPATH}" exec "$REAL_PYTHON" "\$@"
EOF
chmod +x "$NOYAML_BIN/python3"
# Sanity-check the two fixtures themselves before trusting any verdict they
# produce. A shim that silently still imports yaml would make every PyYAML
# assertion below pass for the wrong reason.
if PATH="$NOYAML_BIN:$PATH" python3 -c 'import yaml' 2>/dev/null; then
fail "the no-PyYAML shim does not actually shadow PyYAML — every PyYAML assertion below would be vacuous"
else
pass "fixture check: the no-PyYAML shim makes 'import yaml' fail while python3 still runs"
fi
if PATH="$NOPY_BIN" command -v python3 > /dev/null 2>&1; then
fail "the no-python3 PATH still resolves python3 — every python3 assertion below would be vacuous"
else
pass "fixture check: the no-python3 PATH resolves no python3"
fi
# A minimal, entirely clean subject for each script. The preflight must fire
# before any measurement, so the subject's own content is irrelevant — which is
# exactly what makes a clean one the right choice: nothing else can produce the
# non-zero exit these cases assert.
SUBJECT_SKILL_DIR="$TMPDIR_T/subject/my-skill"
mkdir -p "$SUBJECT_SKILL_DIR"
cat > "$SUBJECT_SKILL_DIR/SKILL.md" <<'EOF'
---
name: my-skill
description: A short valid description. Do not use for anything else.
---
Do the thing.
EOF
SUBJECT_AGENT_ROOT="$TMPDIR_T/subject-agent"
mkdir -p "$SUBJECT_AGENT_ROOT/.apm/agents"
cat > "$SUBJECT_AGENT_ROOT/apm.yml" <<'EOF'
name: test-package
version: 0.1.0
type: skill
EOF
cat > "$SUBJECT_AGENT_ROOT/.apm/agents/my-agent.agent.md" <<'EOF'
---
name: my-agent
description: A short valid description. Do not use for anything else.
---
You are a test agent. When invoked, do the thing.
EOF
# probe_preflight <label> <env-kind: nopython|noyaml> <expect-needle> <cmd...>
probe_preflight() {
local label="$1" kind="$2" needle="$3"
shift 3
local out status=0
set +e
if [[ "$kind" == nopython ]]; then
out="$(env -i PATH="$NOPY_BIN" HOME="$HOME" "$BASH_BIN" "$@" 2>&1)"
else
out="$(env PATH="$NOYAML_BIN:$PATH" "$BASH_BIN" "$@" 2>&1)"
fi
status=$?
set -e
if [[ $status -eq 0 ]]; then
fail "$label exited 0 — a missing hard dependency became a vacuous pass (output: ${out:-<empty>})"
elif [[ "$out" != *"$needle"* ]]; then
# The needle is the DIAGNOSTIC ("python3 is required"), not the bare word.
# Deleting the preflight entirely would still produce a non-zero exit and a
# message mentioning python3 -- bash's own "python3: command not found" --
# so a bare-word needle would go green on a script with no preflight at all.
fail "$label exited $status but never produced the '$needle' diagnostic (output: ${out:-<empty>})"
elif [[ "$kind" == nopython && "$out" == *PyYAML* ]]; then
fail "$label reported PyYAML when python3 itself is missing — that sends the reader to install the wrong thing (output: $out)"
else
pass "$label"
fi
}
echo ""
echo "--- a PATH with no python3 is a hard failure in all three scripts, naming python3 ---"
probe_preflight "scripts/skill-size-check.sh reports missing python3" \
nopython "python3 is required" \
"$HOOK" "$SUBJECT_SKILL_DIR/SKILL.md"
probe_preflight "skill-audit/scripts/validate.sh reports missing python3" \
nopython "python3 is required" \
"$SKILL_VALIDATE" "$SUBJECT_SKILL_DIR"
probe_preflight "agent-audit/scripts/validate.sh reports missing python3" \
nopython "python3 is required" \
"$AGENT_VALIDATE" "$SUBJECT_AGENT_ROOT/.apm/agents/my-agent.agent.md"
echo ""
echo "--- a python3 that cannot import yaml is a hard failure in all three scripts, naming PyYAML ---"
probe_preflight "scripts/skill-size-check.sh reports missing PyYAML" \
noyaml "PyYAML is required" \
"$HOOK" "$SUBJECT_SKILL_DIR/SKILL.md"
probe_preflight "skill-audit/scripts/validate.sh reports missing PyYAML" \
noyaml "PyYAML is required" \
"$SKILL_VALIDATE" "$SUBJECT_SKILL_DIR"
probe_preflight "agent-audit/scripts/validate.sh reports missing PyYAML" \
noyaml "PyYAML is required" \
"$AGENT_VALIDATE" "$SUBJECT_AGENT_ROOT/.apm/agents/my-agent.agent.md"
# The control. Without it, "fails when the dependency is missing" is satisfied by
# a script that fails unconditionally, and the two cases above would be green on
# a gate that never runs at all.
echo ""
echo "--- control: with both dependencies present the same subjects pass ---"
for probe in "$HOOK:$SUBJECT_SKILL_DIR/SKILL.md" \
"$SKILL_VALIDATE:$SUBJECT_SKILL_DIR" \
"$AGENT_VALIDATE:$SUBJECT_AGENT_ROOT/.apm/agents/my-agent.agent.md"; do
script="${probe%%:*}"
arg="${probe#*:}"
set +e
ctl_out="$(bash "$script" "$arg" 2>&1)"
ctl_rc=$?
set -e
if [[ $ctl_rc -eq 0 ]]; then
pass "${script#"$REPO_ROOT/"} exits 0 on a clean subject with python3 and PyYAML available"
else
fail "${script#"$REPO_ROOT/"} failed a clean subject (exit $ctl_rc): $ctl_out"
fi
done
# ---------------------------------------------------------------------------
# 3. verbose: true on the skill-size-check hook, in BOTH manifests
# ---------------------------------------------------------------------------
# .pre-commit-config.yaml governs this repo; .pre-commit-hooks.yaml is what a
# CONSUMER repo gets when it points at this one. Dropping the flag from either
# silences the SUGGESTION tier for that audience alone, which is the hardest
# version of the defect to notice.
echo ""
echo "--- the skill-size-check hook declares verbose: true in both manifests ---"
VERBOSE_REPORT="$(python3 - "$REPO_ROOT" <<'PY'
import os
import sys
import yaml
root = sys.argv[1]
def emit(status, msg):
print("%s\t%s" % (status, msg))
# Repo config: nested repos[].hooks[].
path = os.path.join(root, '.pre-commit-config.yaml')
try:
with open(path, encoding='utf-8') as fh:
cfg = yaml.safe_load(fh) or {}
except Exception as exc:
emit('FAIL', '.pre-commit-config.yaml did not parse: %s' % exc)
cfg = {}
found = None
for repo in cfg.get('repos') or []:
for hook in (repo.get('hooks') or []):
if hook.get('id') == 'skill-size-check':
found = hook
if found is None:
emit('FAIL', '.pre-commit-config.yaml declares no hook with id skill-size-check')
elif found.get('verbose') is True:
emit('PASS', '.pre-commit-config.yaml: skill-size-check is verbose: true')
else:
emit('FAIL', '.pre-commit-config.yaml: skill-size-check has verbose=%r — '
'pre-commit prints nothing for a passing hook, so every '
'ADR-0020 SUGGESTION is swallowed' % (found.get('verbose'),))
# Consumer manifest: a flat list of hooks.
path = os.path.join(root, '.pre-commit-hooks.yaml')
try:
with open(path, encoding='utf-8') as fh:
hooks = yaml.safe_load(fh) or []
except Exception as exc:
emit('FAIL', '.pre-commit-hooks.yaml did not parse: %s' % exc)
hooks = []
found = None
for hook in hooks:
if isinstance(hook, dict) and hook.get('id') == 'kyberforge-skill-size-check':
found = hook
if found is None:
emit('FAIL', '.pre-commit-hooks.yaml declares no hook with id kyberforge-skill-size-check')
elif found.get('verbose') is True:
emit('PASS', '.pre-commit-hooks.yaml: kyberforge-skill-size-check is verbose: true')
else:
emit('FAIL', '.pre-commit-hooks.yaml: kyberforge-skill-size-check has verbose=%r — '
'a consumer repo would never see the SUGGESTION tier'
% (found.get('verbose'),))
PY
)"
while IFS=$'\t' read -r status msg; do
[[ -n "$status" ]] || continue
if [[ "$status" == PASS ]]; then
pass "$msg"
else
fail "$msg"
fi
done <<< "$VERBOSE_REPORT"
echo ""
echo "Results: $PASS passed, $FAIL failed"
[[ $FAIL -eq 0 ]]