refactor(lint): cache manifest parsing, single-pass size check
Two more efficiency findings from the same code-review pass: - check-vale-style-sync.sh's hook_file_regexes() reparsed both pre-commit manifests from scratch on every call. The final validation loop calls it once per probe (3 probes: skill-audit once, agent-audit twice for its two file shapes), so agent-audit's regex set was being parsed twice for no reason. Now cached per skill in a lazily-populated associative array, with a separate "seen" map so an empty result isn't mistaken for "not yet computed." - skill-size-check.sh read the target file twice (separate awk and wc -w calls) to get line and word counts; now a single awk pass returns both. Also documented, next to MAX_LINES/MAX_WORDS, why those constants are duplicated against skill-audit/scripts/ validate.sh's Python implementation rather than unified — same cross-language/cross-context tradeoff as vale-wrap.sh's duplication, guarded by tests/test-skill-size-check.sh's drift check. Verified: test-check-vale-style-sync.sh 20/20, test-skill-size-check.sh 9/9, full suite 12/12, pre-commit --all-files clean.
This commit is contained in:
@@ -80,26 +80,45 @@ done
|
|||||||
# Prints the `files:` regex of every hook, in either manifest, whose entry is
|
# Prints the `files:` regex of every hook, in either manifest, whose entry is
|
||||||
# $1's vale-wrap.sh. Records are delimited by their `- id:` line, so the check
|
# $1's vale-wrap.sh. Records are delimited by their `- id:` line, so the check
|
||||||
# does not depend on `entry:` preceding `files:` within a record.
|
# does not depend on `entry:` preceding `files:` within a record.
|
||||||
|
#
|
||||||
|
# Cached per skill (in HOOK_REGEX_CACHE, populated lazily) because the final
|
||||||
|
# validation loop below probes agent-audit twice — once for its CC agent-file
|
||||||
|
# shape, once for its Copilot .agent.md shape — and both probes need the same
|
||||||
|
# regex set. Without the cache, that pair of calls would each re-parse both
|
||||||
|
# manifest files from scratch for no new information. HOOK_REGEX_CACHE_SEEN is
|
||||||
|
# a separate array so a skill with no matching hooks (empty result) is still
|
||||||
|
# recognized as already computed, rather than re-parsed every call.
|
||||||
|
declare -A HOOK_REGEX_CACHE=()
|
||||||
|
declare -A HOOK_REGEX_CACHE_SEEN=()
|
||||||
hook_file_regexes() {
|
hook_file_regexes() {
|
||||||
local skill="$1" manifest raw
|
local skill="$1" manifest raw result
|
||||||
for manifest in "$REPO_ROOT/.pre-commit-hooks.yaml" "$REPO_ROOT/.pre-commit-config.yaml"; do
|
if [[ -n "${HOOK_REGEX_CACHE_SEEN[$skill]:-}" ]]; then
|
||||||
[[ -f "$manifest" ]] || continue
|
printf '%s' "${HOOK_REGEX_CACHE[$skill]}"
|
||||||
awk -v skill="$skill" '
|
return
|
||||||
function flush() {
|
fi
|
||||||
if (entry ~ skill "/scripts/vale-wrap.sh" && files != "") print files
|
result="$(
|
||||||
entry = ""; files = ""
|
for manifest in "$REPO_ROOT/.pre-commit-hooks.yaml" "$REPO_ROOT/.pre-commit-config.yaml"; do
|
||||||
}
|
[[ -f "$manifest" ]] || continue
|
||||||
/^[ \t]*-[ \t]*id:/ { flush() }
|
awk -v skill="$skill" '
|
||||||
/^[ \t]*entry:/ { entry = $0 }
|
function flush() {
|
||||||
/^[ \t]*files:/ { files = $0; sub(/^[ \t]*files:[ \t]*/, "", files) }
|
if (entry ~ skill "/scripts/vale-wrap.sh" && files != "") print files
|
||||||
END { flush() }
|
entry = ""; files = ""
|
||||||
' "$manifest"
|
}
|
||||||
done | while IFS= read -r raw; do
|
/^[ \t]*-[ \t]*id:/ { flush() }
|
||||||
# Strip the surrounding YAML quotes; the regex itself never carries them.
|
/^[ \t]*entry:/ { entry = $0 }
|
||||||
raw="${raw%\'}"; raw="${raw#\'}"
|
/^[ \t]*files:/ { files = $0; sub(/^[ \t]*files:[ \t]*/, "", files) }
|
||||||
raw="${raw%\"}"; raw="${raw#\"}"
|
END { flush() }
|
||||||
printf '%s\n' "$raw"
|
' "$manifest"
|
||||||
done
|
done | while IFS= read -r raw; do
|
||||||
|
# Strip the surrounding YAML quotes; the regex itself never carries them.
|
||||||
|
raw="${raw%\'}"; raw="${raw#\'}"
|
||||||
|
raw="${raw%\"}"; raw="${raw#\"}"
|
||||||
|
printf '%s\n' "$raw"
|
||||||
|
done
|
||||||
|
)"
|
||||||
|
HOOK_REGEX_CACHE[$skill]="$result"
|
||||||
|
HOOK_REGEX_CACHE_SEEN[$skill]=1
|
||||||
|
printf '%s' "$result"
|
||||||
}
|
}
|
||||||
|
|
||||||
# Asks vale — the thing that actually applies these globs — whether a config
|
# Asks vale — the thing that actually applies these globs — whether a config
|
||||||
|
|||||||
@@ -35,6 +35,13 @@ set -euo pipefail
|
|||||||
# tokenization and does not replace one. Re-measure the corpus before treating
|
# tokenization and does not replace one. Re-measure the corpus before treating
|
||||||
# any of these numbers as still current.
|
# any of these numbers as still current.
|
||||||
|
|
||||||
|
# These constants are intentionally duplicated in
|
||||||
|
# skill-audit/scripts/validate.sh (Python) rather than shared from one file:
|
||||||
|
# this script is a standalone bash pre-commit hook, that one is an in-skill
|
||||||
|
# Python validator invoked in a different context (same rationale as
|
||||||
|
# vale-wrap.sh's per-plugin duplication — see its own header comment).
|
||||||
|
# tests/test-skill-size-check.sh asserts both files agree on these values, so
|
||||||
|
# drift between them fails CI rather than silently diverging.
|
||||||
MAX_LINES=500
|
MAX_LINES=500
|
||||||
MAX_WORDS=2770
|
MAX_WORDS=2770
|
||||||
FAIL=0
|
FAIL=0
|
||||||
@@ -42,16 +49,19 @@ FAIL=0
|
|||||||
for f in "$@"; do
|
for f in "$@"; do
|
||||||
[[ -f "$f" ]] || continue
|
[[ -f "$f" ]] || continue
|
||||||
|
|
||||||
# awk's NR counts the final line even without a trailing newline, matching
|
# Single awk pass computes both line count and word count, avoiding a
|
||||||
# Python's splitlines() semantics (used by skill-audit/scripts/validate.sh
|
# second read of the file. NR counts the final line even without a
|
||||||
# for its own line count) — `wc -l` undercounts by 1 in that case.
|
# trailing newline, matching Python's splitlines() semantics (used by
|
||||||
lines=$(awk 'END{print NR}' "$f")
|
# skill-audit/scripts/validate.sh for its own line count) — `wc -l`
|
||||||
|
# undercounts by 1 in that case. Word count uses awk's default
|
||||||
|
# whitespace-splitting NF, matching `wc -w` semantics.
|
||||||
|
read -r lines words <<< "$(awk '{w += NF} END{print NR, w+0}' "$f")"
|
||||||
|
|
||||||
if (( lines > MAX_LINES )); then
|
if (( lines > MAX_LINES )); then
|
||||||
echo "ERROR: $f has $lines lines, exceeding the $MAX_LINES-line ceiling (agentskills.io skill-authoring.md)" >&2
|
echo "ERROR: $f has $lines lines, exceeding the $MAX_LINES-line ceiling (agentskills.io skill-authoring.md)" >&2
|
||||||
FAIL=1
|
FAIL=1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
words=$(wc -w < "$f")
|
|
||||||
if (( words > MAX_WORDS )); then
|
if (( words > MAX_WORDS )); then
|
||||||
echo "ERROR: $f has $words words (proxy for tokens), exceeding the $MAX_WORDS-word ceiling (~5,000 tokens, agentskills.io skill-authoring.md)" >&2
|
echo "ERROR: $f has $words words (proxy for tokens), exceeding the $MAX_WORDS-word ceiling (~5,000 tokens, agentskills.io skill-authoring.md)" >&2
|
||||||
FAIL=1
|
FAIL=1
|
||||||
|
|||||||
Reference in New Issue
Block a user