refactor(lint): cache manifest parsing, single-pass size check

Two more efficiency findings from the same code-review pass:

- check-vale-style-sync.sh's hook_file_regexes() reparsed both
  pre-commit manifests from scratch on every call. The final
  validation loop calls it once per probe (3 probes: skill-audit once,
  agent-audit twice for its two file shapes), so agent-audit's regex
  set was being parsed twice for no reason. Now cached per skill in a
  lazily-populated associative array, with a separate "seen" map so an
  empty result isn't mistaken for "not yet computed."
- skill-size-check.sh read the target file twice (separate awk and
  wc -w calls) to get line and word counts; now a single awk pass
  returns both. Also documented, next to MAX_LINES/MAX_WORDS, why
  those constants are duplicated against skill-audit/scripts/
  validate.sh's Python implementation rather than unified — same
  cross-language/cross-context tradeoff as vale-wrap.sh's duplication,
  guarded by tests/test-skill-size-check.sh's drift check.

Verified: test-check-vale-style-sync.sh 20/20, test-skill-size-check.sh
9/9, full suite 12/12, pre-commit --all-files clean.
This commit is contained in:
2026-08-09 20:14:36 +00:00
parent 680aa4f43c
commit e62f68a1cc
2 changed files with 53 additions and 24 deletions

View File

@@ -80,26 +80,45 @@ done
# Prints the `files:` regex of every hook, in either manifest, whose entry is # Prints the `files:` regex of every hook, in either manifest, whose entry is
# $1's vale-wrap.sh. Records are delimited by their `- id:` line, so the check # $1's vale-wrap.sh. Records are delimited by their `- id:` line, so the check
# does not depend on `entry:` preceding `files:` within a record. # does not depend on `entry:` preceding `files:` within a record.
#
# Cached per skill (in HOOK_REGEX_CACHE, populated lazily) because the final
# validation loop below probes agent-audit twice — once for its CC agent-file
# shape, once for its Copilot .agent.md shape — and both probes need the same
# regex set. Without the cache, that pair of calls would each re-parse both
# manifest files from scratch for no new information. HOOK_REGEX_CACHE_SEEN is
# a separate array so a skill with no matching hooks (empty result) is still
# recognized as already computed, rather than re-parsed every call.
declare -A HOOK_REGEX_CACHE=()
declare -A HOOK_REGEX_CACHE_SEEN=()
hook_file_regexes() { hook_file_regexes() {
local skill="$1" manifest raw local skill="$1" manifest raw result
for manifest in "$REPO_ROOT/.pre-commit-hooks.yaml" "$REPO_ROOT/.pre-commit-config.yaml"; do if [[ -n "${HOOK_REGEX_CACHE_SEEN[$skill]:-}" ]]; then
[[ -f "$manifest" ]] || continue printf '%s' "${HOOK_REGEX_CACHE[$skill]}"
awk -v skill="$skill" ' return
function flush() { fi
if (entry ~ skill "/scripts/vale-wrap.sh" && files != "") print files result="$(
entry = ""; files = "" for manifest in "$REPO_ROOT/.pre-commit-hooks.yaml" "$REPO_ROOT/.pre-commit-config.yaml"; do
} [[ -f "$manifest" ]] || continue
/^[ \t]*-[ \t]*id:/ { flush() } awk -v skill="$skill" '
/^[ \t]*entry:/ { entry = $0 } function flush() {
/^[ \t]*files:/ { files = $0; sub(/^[ \t]*files:[ \t]*/, "", files) } if (entry ~ skill "/scripts/vale-wrap.sh" && files != "") print files
END { flush() } entry = ""; files = ""
' "$manifest" }
done | while IFS= read -r raw; do /^[ \t]*-[ \t]*id:/ { flush() }
# Strip the surrounding YAML quotes; the regex itself never carries them. /^[ \t]*entry:/ { entry = $0 }
raw="${raw%\'}"; raw="${raw#\'}" /^[ \t]*files:/ { files = $0; sub(/^[ \t]*files:[ \t]*/, "", files) }
raw="${raw%\"}"; raw="${raw#\"}" END { flush() }
printf '%s\n' "$raw" ' "$manifest"
done done | while IFS= read -r raw; do
# Strip the surrounding YAML quotes; the regex itself never carries them.
raw="${raw%\'}"; raw="${raw#\'}"
raw="${raw%\"}"; raw="${raw#\"}"
printf '%s\n' "$raw"
done
)"
HOOK_REGEX_CACHE[$skill]="$result"
HOOK_REGEX_CACHE_SEEN[$skill]=1
printf '%s' "$result"
} }
# Asks vale — the thing that actually applies these globs — whether a config # Asks vale — the thing that actually applies these globs — whether a config

View File

@@ -35,6 +35,13 @@ set -euo pipefail
# tokenization and does not replace one. Re-measure the corpus before treating # tokenization and does not replace one. Re-measure the corpus before treating
# any of these numbers as still current. # any of these numbers as still current.
# These constants are intentionally duplicated in
# skill-audit/scripts/validate.sh (Python) rather than shared from one file:
# this script is a standalone bash pre-commit hook, that one is an in-skill
# Python validator invoked in a different context (same rationale as
# vale-wrap.sh's per-plugin duplication — see its own header comment).
# tests/test-skill-size-check.sh asserts both files agree on these values, so
# drift between them fails CI rather than silently diverging.
MAX_LINES=500 MAX_LINES=500
MAX_WORDS=2770 MAX_WORDS=2770
FAIL=0 FAIL=0
@@ -42,16 +49,19 @@ FAIL=0
for f in "$@"; do for f in "$@"; do
[[ -f "$f" ]] || continue [[ -f "$f" ]] || continue
# awk's NR counts the final line even without a trailing newline, matching # Single awk pass computes both line count and word count, avoiding a
# Python's splitlines() semantics (used by skill-audit/scripts/validate.sh # second read of the file. NR counts the final line even without a
# for its own line count) — `wc -l` undercounts by 1 in that case. # trailing newline, matching Python's splitlines() semantics (used by
lines=$(awk 'END{print NR}' "$f") # skill-audit/scripts/validate.sh for its own line count) — `wc -l`
# undercounts by 1 in that case. Word count uses awk's default
# whitespace-splitting NF, matching `wc -w` semantics.
read -r lines words <<< "$(awk '{w += NF} END{print NR, w+0}' "$f")"
if (( lines > MAX_LINES )); then if (( lines > MAX_LINES )); then
echo "ERROR: $f has $lines lines, exceeding the $MAX_LINES-line ceiling (agentskills.io skill-authoring.md)" >&2 echo "ERROR: $f has $lines lines, exceeding the $MAX_LINES-line ceiling (agentskills.io skill-authoring.md)" >&2
FAIL=1 FAIL=1
fi fi
words=$(wc -w < "$f")
if (( words > MAX_WORDS )); then if (( words > MAX_WORDS )); then
echo "ERROR: $f has $words words (proxy for tokens), exceeding the $MAX_WORDS-word ceiling (~5,000 tokens, agentskills.io skill-authoring.md)" >&2 echo "ERROR: $f has $words words (proxy for tokens), exceeding the $MAX_WORDS-word ceiling (~5,000 tokens, agentskills.io skill-authoring.md)" >&2
FAIL=1 FAIL=1