8 Commits
Author SHA1 Message Date
Defame1297andClaude Haiku 4.5 5e3a1376be chore: fix shellcheck warnings
- Remove unused COLOR_CYAN variable from statusline-command.sh
- Add shellcheck disable directive to deploy-manifest.sh (variables sourced by install.sh)

Co-Authored-By: Claude Haiku 4.5 <[email protected]>
2026-06-27 19:36:25 +00:00
Defame1297 e43fbb4bdf test: verify all hooks working 2026-06-27 19:25:31 +00:00
Defame1297andClaude Haiku 4.5 c0e54a11fa chore: migrate pre-push hook to pre-commit framework
Add pre-push stage hooks:
- run-tests: execute all test-*.sh files and bats suite
- check-manifests: validate marketplace.json and plugin.json paths

Remove hardcoded .git/hooks/pre-push.

Co-Authored-By: Claude Haiku 4.5 <[email protected]>
2026-06-27 19:22:32 +00:00
Defame1297 5da0d34690 chore(pre-commit): activate conventional commits validation hook 2026-06-27 19:20:30 +00:00
Defame1297andClaude Haiku 4.5 5b8b6f529c chore: remove stale setup scripts, skills, and evals
Remove artifacts from pre-commit migration and cleanup:
- scripts/setup-gitleaks.sh, scripts/gitleaks.toml: legacy setup scripts
- tests/test-setup-gitleaks.sh, tests/test-setup-hooks.sh: phantom test files
- plugins/bin/skills/gitleaks/: skill for deprecated shell-based setup
- plugins/bin/evals/cross-cutting/gitleaks/: eval for deleted skill
- plugins/bin/evals/cross-cutting/neuledge-context/: out-of-scope eval
- plugins/bin/evals/implement/write-docs/: incomplete eval
- package.json, package-lock.json: markdownlint dependencies (unused)

All validation now managed by .pre-commit-config.yaml.

Co-Authored-By: Claude Haiku 4.5 <[email protected]>
2026-06-27 19:04:53 +00:00
Defame1297andClaude Haiku 4.5 4cbc993af4 docs: remove stale references to deleted setup scripts
Update docs to reflect pre-commit migration and cleanup:
- spec/overview.md: removed phantom test file references
- ROADMAP.md: removed references to non-existent test files
- LESSONS.md: removed reference to setup-hooks.sh bug

Co-Authored-By: Claude Haiku 4.5 <[email protected]>
2026-06-27 19:04:20 +00:00
Defame1297andClaude Haiku 4.5 a18b7b46ac chore: remove legacy pre-commit setup scripts
Removed .git/hooks/pre-commit.legacy and scripts/setup-hooks.sh — fully replaced by .pre-commit-config.yaml.

Co-Authored-By: Claude Haiku 4.5 <[email protected]>
2026-06-27 18:55:58 +00:00
Defame1297andClaude Haiku 4.5 d98d0dae18 chore: migrate legacy pre-commit hook to .pre-commit-config.yaml
Replaces shell script (.git/hooks/pre-commit.legacy) with ecosystem-managed pre-commit framework:
- gitleaks/gitleaks: secret scanning
- jumanjihouse/pre-commit-hooks: shellcheck wrapper
- pre-commit/pre-commit-hooks: JSON/YAML validation, end-of-file-fixer, trailing-whitespace
- local hooks: SKILL.md frontmatter validation

Uses pinned versions for reproducibility across environments. Includes auto-fixes from hook runs (formatting, trailing whitespace, JSON beautification).

Co-Authored-By: Claude Haiku 4.5 <[email protected]>
2026-06-27 18:54:37 +00:00
53 changed files with 262 additions and 1415 deletions

No files matched your search

-3
View File
@@ -24,6 +24,3 @@ node_modules/
# Claude Code local settings (machine-specific)
.claude/settings.local.json
graphify-out/cost.json # local only
graphify-out/cache/ # optional: commit for speed, skip to keep repo small
+1 -1
View File
@@ -9,4 +9,4 @@
]
}
}
}
}
+72
View File
@@ -0,0 +1,72 @@
repos:
- repo: https://github.com/compilerla/conventional-pre-commit
rev: v2.4.0
hooks:
- id: conventional-pre-commit
stages: [commit-msg]
- repo: https://github.com/gitleaks/gitleaks
rev: v8.21.2
hooks:
- id: gitleaks
stages: ['pre-commit']
- repo: https://github.com/jumanjihouse/pre-commit-hooks
rev: 3.0.0
hooks:
- id: shellcheck
args: [--severity=warning]
stages: ['pre-commit']
- repo: https://github.com/pre-commit/pre-commit-hooks
rev: v4.5.0
hooks:
- id: end-of-file-fixer
stages: ['pre-commit']
- id: check-json
stages: ['pre-commit']
- id: pretty-format-json
stages: ['pre-commit']
- id: check-yaml
stages: ['pre-commit']
- id: trailing-whitespace
stages: ['pre-commit']
- repo: local
hooks:
- id: run-tests
name: Run test suite
description: Run all test-*.sh files and bats suite
entry: bash tests/run-tests.sh
language: system
stages: [pre-push]
pass_filenames: false
always_run: true
- id: check-manifests
name: Check plugin manifests
description: Validate marketplace.json and plugin.json paths
entry: bash scripts/check-manifests.sh
language: system
stages: [pre-push]
pass_filenames: false
always_run: true
- id: skill-frontmatter
stages: ['pre-commit']
name: SKILL.md frontmatter validation
description: Ensure SKILL.md files have required frontmatter fields
entry: bash
language: system
files: 'SKILL\.md$'
args:
- -c
- |
for f in "$@"; do
if [[ -f "$f" ]]; then
if ! grep -q "^name:" "$f" || ! grep -q "^description:" "$f"; then
echo "ERROR: $f is missing required frontmatter fields (name: and description:)"
exit 1
fi
fi
done
+6 -2
View File
@@ -94,9 +94,13 @@ When running a full test audit, `claude plugin validate --strict` was not includ
`scripts/gitleaks.toml` (source, in git, deployed to repo root by `setup-gitleaks.sh`) and `.gitleaks.toml` (deployed root copy, read by the hook, also tracked in git) were found with different allowlist states — someone had updated the deployed file directly without updating the source. Running `setup-gitleaks.sh` again would overwrite the deployed file with the stale source, silently deleting the existing allowlist and re-exposing a known false positive as a blocking pre-commit failure. Fix: treat `scripts/gitleaks.toml` as the single source of truth; never edit `.gitleaks.toml` directly. When making allowlist changes, always update source and deployed copy together in the same commit. Longer-term fix: `setup-gitleaks.sh` should merge rather than overwrite, or detect divergence and warn when `.gitleaks.toml` is tracked in git.
## 2026-06-21 — `shellcheck` without `-x` blocks pre-commit on any script using `source`
## 2026-06-21 — `shellcheck` without `-x` blocks pre-commit on any script using `source` (LEGACY SHELL HOOKS)
The pre-commit hook ran `shellcheck "$f"` without `-x`. Without `-x`, shellcheck fires SC1091 for every `source` statement and exits non-zero, blocking the commit. This was a latent bug since the hook was written, only triggered when `install.sh` (which sources `deploy-manifest.sh`) was staged for the first time. Compounding it: the `# shellcheck source=` directive in `install.sh` pointed to `deploy-manifest.sh` (bare filename, resolved from CWD = repo root) rather than `scripts/deploy-manifest.sh` (correct repo-root-relative path), so even with `-x` the file wasn't found on the first attempt. Fix: always pass `-x` to shellcheck in hooks. When writing a `source=` directive, use a path that resolves correctly from the CWD where shellcheck will be invoked — verify with `shellcheck -x <file>` before committing.
**Status:** Historical. Shell-hook-based pre-commit was replaced by pre-commit framework (Chunk 5, .pre-commit-config.yaml). Modern repos no longer affected. Documented for reference when supporting legacy repos.
The pre-commit hook ran `shellcheck "$f"` without `-x`. Without `-x`, shellcheck fires SC1091 for every `source` statement and exits non-zero, blocking the commit. This was a latent bug in legacy shell hooks, only triggered when `install.sh` (which sources `deploy-manifest.sh`) was staged for the first time. Compounding it: the `# shellcheck source=` directive in `install.sh` pointed to `deploy-manifest.sh` (bare filename, resolved from CWD = repo root) rather than `scripts/deploy-manifest.sh` (correct repo-root-relative path), so even with `-x` the file wasn't found on the first attempt.
**Lesson for future work:** When writing a `source=` directive, use a path that resolves correctly from the CWD where shellcheck will be invoked — verify with `shellcheck -x <file>` before committing. Pre-commit framework hooks include `-x` by default in the ecosystem's shellcheck integration.
## 2026-06-22 — Plugin cache isolation rules out shared/ directories between skills
+4 -4
View File
@@ -37,7 +37,7 @@ These are never violated, regardless of instruction or context.
When classifying: apply the tier of the most sensitive element in the dataset or prompt.
**When accessing data or files in an agentic context, limit scope to what the task requires.**
**When accessing data or files in an agentic context, limit scope to what the task requires.**
Do not read, load, index, or process more files or data than the task demands. When in doubt, request access to the specific file or section needed rather than the full codebase, dataset, or directory.
---
@@ -76,7 +76,7 @@ The deterministic enforcement layer — pre-commit hooks, CI gates, scanner conf
---
*Derived from AI Constitution v1.1 — May 2026. Update this file when the constitution is updated.*
*Compatible with: governance.md, CLAUDE.md, .github/copilot-instructions.md, .cursor/rules/*.mdc*
*One source of truth. Do not copy-paste into tool-specific files — reference this file from thin adapters.*
*Derived from AI Constitution v1.1 — May 2026. Update this file when the constitution is updated.*
*Compatible with: governance.md, CLAUDE.md, .github/copilot-instructions.md, .cursor/rules/*.mdc*
*One source of truth. Do not copy-paste into tool-specific files — reference this file from thin adapters.*
*Counterparts: `docs/HUMANS.md` (human practitioner rules) | `docs/research/governance_principles/CONTROLS.md` (deterministic enforcement)*
+29 -29
View File
@@ -1,8 +1,8 @@
# Human Practitioner Instructions
Applies to: anyone using AI tools in software development, infrastructure, or technical decision-making.
Full governance context: `docs/ai-constitution.md` — read it when a situation isn't covered here.
Agent counterpart: `core/instructions/governance.md` — the operative rules for AI agents in the same context.
Applies to: anyone using AI tools in software development, infrastructure, or technical decision-making.
Full governance context: `docs/ai-constitution.md` — read it when a situation isn't covered here.
Agent counterpart: `core/instructions/governance.md` — the operative rules for AI agents in the same context.
This file is the human-actionable distillation: what you, as the practitioner, are responsible for.
---
@@ -21,89 +21,89 @@ These are never compromised, regardless of deadline, convenience, or context.
## Before: Starting an AI-Assisted Task
**Classify the data you're about to share.**
**Classify the data you're about to share.**
Ask: what tier is this? Public, Internal, Confidential, or Restricted? Apply the tier of the most sensitive element. If it's Confidential, confirm you're using a tool with contractual data-not-trained guarantees. If it's Restricted, stop — it doesn't enter AI context.
**Send only what the task requires.**
**Send only what the task requires.**
Do not share full codebases, entire logs, or complete datasets when a relevant excerpt would serve equally well. Anonymise or pseudonymise personal data before AI input wherever feasible. More context than necessary increases exposure without improving the output.
**Use the right tool for the data tier.**
**Use the right tool for the data tier.**
Consumer and free-tier AI products handle Public data only. Everything else requires enterprise tooling with an explicit contractual commitment. Verify per provider; do not assume.
**Define what success looks like before you start.**
**Define what success looks like before you start.**
AI usage without a success criterion is unjustifiable — the environmental and operational costs are real. What does a good outcome look like? How will you know if the AI helped or misled you?
**Know what scope you're granting.**
**Know what scope you're granting.**
If you're running an agentic workflow, be explicit about what the agent may and may not do before it starts. Ambiguous scope means the agent will make judgment calls you didn't authorise.
---
## During: Working with the AI
**Don't trust confident output — especially fluent, well-formatted confident output.**
**Don't trust confident output — especially fluent, well-formatted confident output.**
Linguistic fluency and factual accuracy are unrelated. Confident language is a sycophancy signal. The more certain and complete an AI response sounds, the more carefully you should validate it.
**On high-stakes questions, don't prompt for brevity.**
**On high-stakes questions, don't prompt for brevity.**
Conciseness instructions demonstrably degrade factual reliability. Where accuracy matters, prompt for accuracy. Ask the AI to show its reasoning.
**On contested, values-laden, or complex technical questions, prompt explicitly for dissenting views.**
**On contested, values-laden, or complex technical questions, prompt explicitly for dissenting views.**
AI outputs are majority-weighted, not neutral. A single response on an architectural decision, risk assessment, or ethical question reflects the dominant training-data perspective. Ask: "What are the strongest arguments against this?" before treating the first output as balanced.
**Cross-validate any output that informs a consequential decision.**
**Cross-validate any output that informs a consequential decision.**
Architecture, security configuration, deployment, legal, financial — validate against an independent source or a second model. AI agreement with itself is not validation.
**Review AI-generated code before accepting it.**
**Review AI-generated code before accepting it.**
Check specifically for: hardcoded credentials; insecure patterns (injection vulnerabilities, overly permissive access); copyleft-licensed fragments (GPL, AGPL) without licence headers; missing or incorrect dependencies. This review is not optional and is not the AI's job.
**Apply a human checkpoint before any production, architecture, or infrastructure change.**
**Apply a human checkpoint before any production, architecture, or infrastructure change.**
No AI-initiated change to production systems, security configuration, or infrastructure is applied without explicit human review and approval of the specific change. This is a hard rule, not a guideline.
**For repeatable tasks, ask AI to generate a script — not to do the task repeatedly.**
**For repeatable tasks, ask AI to generate a script — not to do the task repeatedly.**
If a task has a correct answer that does not depend on context or judgement, use AI once to write a script that runs it deterministically. The script goes in version control; the script is the governed artefact. Invoking AI inference each time a repeatable task runs adds cost, unreliability, and attack surface for no benefit. The break-even is roughly 17 invocations — anything recurring beyond that should be codified.
**Manage the volume of AI-generated output to what you can genuinely evaluate.**
When an agentic workflow generates large quantities of code or changes, approving them as a batch is not review — it is rubber-stamping. If throughput exceeds your verification capacity, reduce it. Output volume is a governance variable, not just a productivity one.
**Manage the volume of AI-generated output to what you can genuinely evaluate.**
When an agentic workflow generates large quantities of code or changes, approving them as a batch is not review — it is rubber-stamping. If throughput exceeds your verification capacity, reduce it. Output volume is a governance variable, not just a productivity one.
Over-reliance on AI for tasks that build critical skills creates cognitive dependency — measurably. If you couldn't do this task without AI and that matters for your ability to audit, debug, or override the AI, that's a governance risk, not just a personal one. Rotate AI-free approaches periodically on skill-critical work.
---
## After: Completing AI-Assisted Work
**Verify you own the output.**
**Verify you own the output.**
Before committing AI-generated code: can you explain what it does and why? Can you modify it at the intent and architecture level? Can you verify its behaviour? If not, you have not reviewed it — you have approved it. These are not the same thing.
**Licence-scan AI-generated code before committing.**
**Licence-scan AI-generated code before committing.**
Copyleft-licensed fragments can appear in AI output without licence headers. Manifest-based scanners don't catch them. Run a dedicated licence scan on AI-assisted contributions.
**Document your human contribution.**
**Document your human contribution.**
Version control history, code review records, and prompt logs together constitute evidence of authorship and accountability. Where IP protection or accountability matters, the human contribution must be substantive and traceable.
**Disclose AI involvement where it affects others.**
**Disclose AI involvement where it affects others.**
If an AI-assisted output informs a decision that affects other people — a report, recommendation, architecture review, or policy — disclose the AI involvement. This is an ethical obligation regardless of legal requirement.
**Log AI-agent actions that produce effects.**
**Log AI-agent actions that produce effects.**
Any agent action that changes state must leave a human-readable trace: what was the prompt, what model, what action was taken, what was the outcome. Isolated timestamps are not sufficient.
**Version prompts used in production.**
**Version prompts used in production.**
Production prompts are code. They need version control, a change log recording what changed and why, and human review before deployment. Unversioned prompts are unauditable.
**If using AI output commercially, verify the provider's IP terms.**
**If using AI output commercially, verify the provider's IP terms.**
Rights to AI-generated outputs vary significantly by provider and tier. Review the terms of service specifically for output ownership clauses, IP indemnification, and restrictions before using AI-assisted code or content in commercial software. Enterprise agreements must address these explicitly — do not assume standard terms provide coverage.
**Measure value delivered.**
**Measure value delivered.**
Did this AI integration do what it was supposed to do? If you defined success before you started, check it now. Deployments that haven't crossed into measurable value delivery must be time-bounded and reviewed, not left running indefinitely.
---
## When Things Go Wrong
**Diagnose first; remediate with human approval.**
**Diagnose first; remediate with human approval.**
AI-assisted diagnosis and root cause analysis can run. Applying remediation to production — rollback, config change, scaling decision — requires explicit human approval unless the action is pre-defined, bounded, and reversible.
**Post-mortem every AI-involved incident.**
**Post-mortem every AI-involved incident.**
Cover: what instructions the agent operated under, what decision it made, what the failure mode was, and what governance change prevents recurrence. AI incidents are not a different category from service incidents — same rigour applies.
**Regulatory notification obligations don't pause because AI was involved.**
**Regulatory notification obligations don't pause because AI was involved.**
GDPR Article 33/34 timelines and thresholds apply regardless of whether an AI system caused or contributed to the incident.
---
@@ -116,5 +116,5 @@ Controls that run mechanically — pre-commit hooks, CI gates, scanner configura
---
*Derived from AI Constitution v1.1 — May 2026.*
*Derived from AI Constitution v1.1 — May 2026.*
*Counterpart to: `core/instructions/governance.md` (agent rules) | `docs/research/governance_principles/CONTROLS.md` (deterministic enforcement) | Full context: `docs/ai-constitution.md`*
+2 -3
View File
@@ -44,11 +44,10 @@ A parallel workstream (not a numbered chunk) that runs alongside the chunk seque
**Pre-Chunk 6 test suite work** (no CI required — can be done now; see Gitea issue #2 for full context):
- Fix U1 first: add gitleaks.toml allowlist entry for `docs/research/ai-coding-factory/ai-coding-factory-session.md:90` (`Token routing: Haiku/Sonnet/Opus` triggers `generic-api-key` false positive; pre-commit hook blocks commits on all machines with gitleaks installed)
- `tests/test-plugin-validate.sh` — run `claude plugin validate --strict` on all plugins and marketplace manifests; add the same check to the pre-push hook alongside `check-manifests.sh`
- `tests/test-hook-integrity.sh` — verify `.git/hooks/pre-commit` is installed, executable, and contains the expected idempotency markers; distinct from `test-setup-hooks.sh` which tests the setup script, not the installed artifact
- `tests/test-gitleaks-scan.sh` — run `gitleaks detect` against the repo and assert exit 0; validates `gitleaks.toml` allowlist correctly suppresses known false positives (requires U1 fix first)
- `tests/test-pre-commit-installed.sh` — verify `.pre-commit-config.yaml` is present and hooks run successfully; validate pre-commit framework integration
- `tests/test-inventory-crossrefs.sh` — run `inventory.sh` against the live repo; assert zero `../` cross-reference warnings in post-refactor skills; triage the 11 current warnings (U4: determine which are in Chunk 3 rebuild targets vs. post-refactor skills that should be self-contained)
- `tests/run-all-tests.sh` — single entry point that runs every test in `tests/`; needed for both developer use and future CI integration
- Extend `test-governance-layer.sh` — add structural checks that CONTROLS.md-required controls are in place (gitleaks wired to pre-commit, hook executable, plugin validation passes strict mode); current checks verify governance files exist but not that controls are enforced
- Extend `test-governance-layer.sh` — add structural checks that CONTROLS.md-required controls are in place (.pre-commit-config.yaml present with gitleaks hook, pre-commit framework installed, plugin validation passes strict mode); current checks verify governance files exist but not that controls are enforced
**Chunk 6 CI gaps** (require CI pipeline; implement during Chunk 6 grill):
- Secret scanning in CI — CONTROLS.md: "Pre-commit hooks can be bypassed; CI cannot. Both layers are required."
+62 -62
View File
@@ -1,52 +1,52 @@
# AI Constitution
**Version:** 1.1 (corrections from deep research pass applied May 2026)
**Scope:** All AI-assisted software development, deployment, and infrastructure management
**Audience:** Humans and AI agents operating in this context
**Inheritance:** Solo-authored; designed to be inherited by future collaborators and AI agents without requiring the author present
**Derivation:** Derived from sourced research across ten governance topics. Principles are evidence-based, not aspirational.
**Version:** 1.1 (corrections from deep research pass applied May 2026)
**Scope:** All AI-assisted software development, deployment, and infrastructure management
**Audience:** Humans and AI agents operating in this context
**Inheritance:** Solo-authored; designed to be inherited by future collaborators and AI agents without requiring the author present
**Derivation:** Derived from sourced research across ten governance topics. Principles are evidence-based, not aspirational.
**Operative agent instructions:** See `core/instructions/governance.md` — the concise, agent-actionable distillation of this document for global context use.
---
## 1. Accountability
**Accountability is non-transferable.**
**Accountability is non-transferable.**
Every AI-generated output that enters a system, codebase, or production environment is owned by the human who accepted it. AI assistance does not reduce or distribute responsibility. "The model produced it" is not a defence — legally, ethically, or operationally.
**Ethics commitments must be concrete and auditable.**
**Ethics commitments must be concrete and auditable.**
Any principle in this document that cannot be tested or verified is not a principle — it is a claim. If compliance cannot be demonstrated, the commitment does not exist.
---
## 2. Security
**Secrets must never enter AI context.**
**Secrets must never enter AI context.**
Credentials, API keys, tokens, passwords, and certificates must not appear in prompts, context files, RAG pipelines, or any input to an AI system. This is an architectural constraint, not a reminder. Scan context before it reaches a model.
**Never use AI-generated secrets, passwords, or cryptographic material.**
**Never use AI-generated secrets, passwords, or cryptographic material.**
LLM-generated passwords have demonstrably insufficient entropy and exhibit predictable patterns. Use cryptographically secure random sources for all credential generation.
**AI-generated code is untrusted by default.**
**AI-generated code is untrusted by default.**
Review AI-generated code with more scrutiny than human-written code — specifically for hardcoded credentials, insecure patterns, and licence-encumbered fragments — before any commit.
**Apply least-privilege to all AI agents.**
**Apply least-privilege to all AI agents.**
Agents receive only the permissions required for their specific, current task. Long-lived, broad-scope tokens for AI agents are prohibited. Scope credentials tightly; rotate frequently.
**Apply OWASP LLM Top 10 and Agentic AI Top 10 as baseline security requirements.**
**Apply OWASP LLM Top 10 and Agentic AI Top 10 as baseline security requirements.**
Prompt injection, supply chain risks, excessive agency, sensitive information disclosure, and system prompt leakage require explicit controls. Traditional AppSec frameworks do not cover these attack surfaces.
**AI pipelines must surface uncertainty; never treat confident AI output as accurate output.**
**AI pipelines must surface uncertainty; never treat confident AI output as accurate output.**
Chaining AI subsystems without propagating confidence levels creates compounding, invisible error. Uncertain outputs require human review before consequential action.
---
## 3. Data Protection & Classification
**Sending personal data to an AI system is data processing under GDPR.**
**Sending personal data to an AI system is data processing under GDPR.**
It requires a lawful basis, a defined purpose, and appropriate safeguards. This applies to prompts, RAG pipelines, and fine-tuning data equally. There is no "just testing" exemption.
**The context window is a data store. Classify it accordingly.**
**The context window is a data store. Classify it accordingly.**
Everything that enters an AI prompt is subject to the same classification obligations as any other data store. Apply the classification framework below.
### Data Classification for AI Systems
@@ -58,168 +58,168 @@ Everything that enters an AI prompt is subject to the same classification obliga
| 3 | **Confidential** | Proprietary source code, system architecture, IP, identifiable personal data | Enterprise AI with explicit data-not-used-for-training contractual commitment; GDPR legal basis required for personal data |
| 4 | **Restricted** | GDPR Article 9 special categories (health, biometrics, ethnicity, religion, sexual orientation, political views), credentials, regulated financial data, data under professional secrecy | Never enters any AI context. Hard architectural prohibition. |
**Consumer and free-tier AI products are incompatible with processing organisational or personal data.**
**Consumer and free-tier AI products are incompatible with processing organisational or personal data.**
Enterprise contracts with explicit data-not-used-for-training commitments are the minimum bar. Verify per provider; do not assume.
**Data minimisation applies to AI prompts.**
**Data minimisation applies to AI prompts.**
Send only what is necessary for the task. Anonymise or pseudonymise personal data before AI input wherever feasible.
**Personal data must not enter AI fine-tuning or RAG pipelines without a GDPR legal basis and a completed DPIA.**
**Personal data must not enter AI fine-tuning or RAG pipelines without a GDPR legal basis and a completed DPIA.**
Right-to-erasure obligations under Article 17 cannot be fulfilled once data is encoded in model weights. This decision is irreversible.
---
## 4. Behaviour & Sycophancy
**Sycophancy is a first-class reliability and ethical risk.**
**Sycophancy is a first-class reliability and ethical risk.**
AI systems trained via RLHF systematically prioritise approval over accuracy. This is the most tractable cause of hallucination and must be explicitly designed against — through prompting standards, model selection, and evaluation criteria.
**Never interpret AI agreement as AI accuracy.**
**Never interpret AI agreement as AI accuracy.**
Models change correct answers to wrong ones under user pressure in a majority of observed cases, then persist in the wrong answer. Challenge AI outputs before trusting them; agreement is not confirmation.
**In high-stakes contexts, never prompt for brevity at the expense of accuracy.**
**In high-stakes contexts, never prompt for brevity at the expense of accuracy.**
Conciseness instructions demonstrably degrade factual reliability. Where accuracy matters, prompt for accuracy.
**Cross-validate consequential AI outputs.**
**Cross-validate consequential AI outputs.**
Any AI-generated output that informs a significant decision — architecture, security configuration, deployment, legal or financial — must be validated against an independent source or a second model before acting on it.
**Select models partly on sycophancy resistance.**
**Select models partly on sycophancy resistance.**
Model selection for professional use must include evaluation of sycophancy behaviour alongside capability benchmarks. Use a portfolio of benchmarks (MASK, SYCON-Bench, SycEval) — rankings flip across evaluations and no single benchmark is reliable. Run your own deployment-stage test for your specific task context; do not rely on vendor or single-study claims about which model family is most resistant.
**In domains where diverse perspectives matter, prompt explicitly for multiple viewpoints and dissenting positions.**
**In domains where diverse perspectives matter, prompt explicitly for multiple viewpoints and dissenting positions.**
AI systems are trained in ways that systematically suppress annotator disagreements, producing outputs weighted toward dominant viewpoints at the expense of minority or dissenting positions (arxiv 2505.07772). A single AI output on a contested, values-laden, or socially complex question is not a neutral summary — it is a majority-weighted perspective. In architecture decisions, risk assessments, ethical questions, and any domain with genuine expert disagreement, prompt for counterarguments and dissenting views explicitly; do not treat the first output as balanced.
**In domains where diverse perspectives matter, prompt explicitly for dissent.**
**In domains where diverse perspectives matter, prompt explicitly for dissent.**
AI systems trained to suppress annotator disagreement produce outputs that systematically underrepresent non-dominant viewpoints (arxiv 2505.07772). In architecture decisions, ethics reviews, risk assessments, and anything affecting underrepresented groups — explicitly prompt for minority positions, dissenting analysis, and counterarguments. Cross-validation against independent sources partially compensates for homogenisation; active prompting for dissent addresses it more directly.
---
## 5. Human Oversight & Automation Boundaries
**Human oversight must be genuine, not symbolic.**
**Human oversight must be genuine, not symbolic.**
Assigning a reviewer does not constitute oversight unless they have the information, time, agency, and intent to evaluate the output meaningfully. Review processes must make genuine evaluation possible.
**Production systems require a human checkpoint before any AI-initiated change.**
**Production systems require a human checkpoint before any AI-initiated change.**
This is a hard rule. No architecture change, infrastructure modification, security configuration, or production deployment may be applied by an AI agent without explicit human review and approval of the specific change.
**Humans must own the code — not just approve it.**
**Humans must own the code — not just approve it.**
The required comprehension standard (ACM/IEEE-CS Software Engineering Code of Ethics) is: intent-level understanding of what the code does and why; architectural understanding of how it fits the system; and verifiable behaviour via tests or traceable reasoning. Line-by-line comprehension of every implementation detail is not required and not the professional standard. What is required: a developer cannot commit AI-generated code they cannot explain, modify at the intent-and-architecture level, or verify against defined behaviour — with or without AI assistance for the verification step itself.
**Limit AI output volume to what reviewers can genuinely evaluate.**
**Limit AI output volume to what reviewers can genuinely evaluate.**
When AI-generated change throughput exceeds human verification capacity, approvals become rubber-stamps. Output rates must be managed to preserve the possibility of genuine review.
**Distinguish HITL from HOTL deliberately.**
**Distinguish HITL from HOTL deliberately.**
Human-in-the-loop (HITL) pauses before consequential action. Human-on-the-loop (HOTL) monitors after the fact. HITL is required for irreversible or high-stakes actions. HOTL is acceptable for low-stakes, bounded, reversible actions. The distinction must be explicit and documented.
**AI assistance must augment human capability, not replace it.**
**AI assistance must augment human capability, not replace it.**
Over-reliance on AI for tasks that require and develop critical skills is a governance risk, not just a quality risk. Kosmyna et al. (2025) found measurable neural disengagement in AI-assisted work; domain evidence shows skill atrophy when AI support is removed; ACM FAccT 2026 identifies cognitive offloading as a systematically overlooked safety risk. When AI takes over a capability entirely, the human's ability to catch AI errors in that domain is also lost. Governance must include periodic assessment of whether AI-assisted roles retain the baseline capability required to operate, audit, and override the AI without it.
**AI assistance must augment human capability, not replace it.**
**AI assistance must augment human capability, not replace it.**
Over-reliance on AI for tasks that require critical thinking, system comprehension, or skilled judgement creates cognitive dependency that degrades organisational resilience over time (Kosmyna et al. 2025; Chalkidis & Søgaard, ACM FAccT 2026). Governance must include mechanisms to detect skill atrophy in AI-assisted roles — periodic AI-free practice, comprehension checks, and capability baselines that do not depend on AI availability.
---
## 6. Sustainability & Societal Cost
**Governance is an obligation to those who bear the costs, not just those who use the tools.**
**Governance is an obligation to those who bear the costs, not just those who use the tools.**
AI's primary costs — environmental, epistemic, and distributional — fall predominantly on people who are not its users: communities bearing grid and water stress from data centres, workers displaced faster than they can upskill, and societies absorbing the epistemic effects of large-scale AI-generated content at scale (IEA Energy and AI 2025; de Vries-Gao, ScienceDirect 2025; Chalkidis & Søgaard, ACM FAccT 2026). Those who benefit from AI use have an obligation to those who bear its costs — whether or not those costs are currently priced or legally required to be accounted for.
**Unmeasured AI usage is unjustifiable.**
**Unmeasured AI usage is unjustifiable.**
Every AI integration must have defined success metrics before deployment. The environmental and societal costs are real and externally borne; they cannot be justified without evidence of value delivered. 42% of enterprises have abandoned most AI initiatives; only 5% of GenAI pilots show measurable P&L impact (S&P Global n=1,006; MIT NANDA lab). If value cannot be articulated, the costs on others cannot be defended.
**Match model capability to task complexity.**
**Match model capability to task complexity.**
Using frontier models for tasks a smaller model handles is not just economically wasteful — it imposes unnecessary environmental and infrastructure costs on others. Model selection is a governance decision with externalities.
**Token efficiency is a sustainability metric, not just a cost metric.**
**Token efficiency is a sustainability metric, not just a cost metric.**
Tokens per unit of value delivered simultaneously tracks cost, carbon intensity, and whether AI is doing genuine work. Per-task energy use is falling rapidly; aggregate consumption rises faster because adoption scale outpaces efficiency gains — the Jevons paradox applied to AI (IEA 2025/2026).
**Apply the J-Curve honestly.**
**Apply the J-Curve honestly.**
AI deployments not yet delivering measurable value must be time-bounded. DORA 2025 confirms the J-Curve pattern: short-term costs precede long-term gains, but the curve must actually turn. If a deployment has not reached value delivery within a defined review period, it must be redesigned or discontinued.
**Treat provider sustainability claims sceptically.**
**Treat provider sustainability claims sceptically.**
Corporate environmental disclosure does not currently distinguish AI from non-AI workloads; independent verification of AI-specific footprint is not possible without regulatory mandates. Source claims only from independently verifiable data (IEA, peer-reviewed studies).
---
## 7. Transparency & Auditability
**Every AI agent action that produces an effect must generate a tamper-evident, human-readable trace.**
**Every AI agent action that produces an effect must generate a tamper-evident, human-readable trace.**
Minimum content: prompt input, model version, output, tool invocations, actor identity, timestamp. Isolated timestamps are not sufficient.
**Prompts are code and must be versioned accordingly.**
**Prompts are code and must be versioned accordingly.**
Every prompt used in a production AI system must be under version control with change logs recording what changed, why, and who approved the change. Unversioned prompts are unauditable prompts.
**AI involvement must be disclosed to anyone affected by its outputs.**
**AI involvement must be disclosed to anyone affected by its outputs.**
This is an ethical obligation regardless of jurisdiction. Under the EU AI Act (post-Omnibus May 2026 agreement): Article 50 transparency obligations apply from **December 2, 2026**, and only to providers of certain AI system types (chatbots, deepfake generators, high-risk systems) — not to deployers using coding assistants internally. Developers using tools like Copilot, Claude Code, or Cursor currently face only **Article 4 (AI literacy)** obligations, which have been live since February 2025. Consult legal counsel for jurisdiction-specific obligations.
**Logging must not create new data protection exposures.**
**Logging must not create new data protection exposures.**
PII in logs must be redacted at ingestion. Log retention periods must align with data protection obligations — retain only what is necessary for the defined audit purpose.
---
## 8. Intellectual Property
**AI-generated code without meaningful human authorship is unprotectable and simultaneously liable.**
**AI-generated code without meaningful human authorship is unprotectable and simultaneously liable.**
It may infringe third-party IP while being ineligible for copyright protection itself. Substantial human review, editing, and integration is required for both IP protection and licence compliance.
**Run licence-scanning on all AI-generated code before committing.**
**Run licence-scanning on all AI-generated code before committing.**
Copyleft-licensed fragments can appear in AI output without licence headers. Manifest-based scanning tools do not catch AI-generated code. Dedicated licence scanning must cover AI-assisted contributions explicitly.
**Review AI provider terms of service specifically for IP provisions.**
**Review AI provider terms of service specifically for IP provisions.**
Rights to AI-generated outputs vary significantly by provider and tier. Enterprise agreements must be reviewed for IP indemnification, output ownership clauses, and restrictions before using AI output in commercial software.
**Document human contributions to AI-assisted code.**
**Document human contributions to AI-assisted code.**
Version control history, code review records, and prompt logs together constitute evidence of human authorship. Where IP protection matters, the human contribution must be substantive and documentable.
---
## 9. Incident Response
**Extend existing IR frameworks for AI-specific failure modes; do not replace them.**
**Extend existing IR frameworks for AI-specific failure modes; do not replace them.**
NIST SP 800-61 and ISO/IEC 27035 remain the required foundation. Extend with specific playbooks covering: prompt injection attacks, agentic scope violations, AI-caused data exposure, and auditability failures. Each requires a distinct detection and response procedure.
**Design for error containment, not error prevention.**
**Design for error containment, not error prevention.**
AI systems will produce erroneous outputs. The primary design obligation is to prevent errors from propagating to consequential, irreversible action — through permission envelopes, scope constraints, and HITL gates.
**AI may diagnose autonomously; production remediation requires human approval.**
**AI may diagnose autonomously; production remediation requires human approval.**
AI-assisted detection and root cause analysis can run without human intervention. Applying remediation to production systems — rollback, configuration change, scaling decision — requires explicit human approval unless the action is pre-defined, bounded, and reversible.
**Post-mortems must cover AI and automation failures explicitly.**
**Post-mortems must cover AI and automation failures explicitly.**
Every AI-involved incident must be post-mortemed with the same rigour as service outages. The post-mortem must address: what instructions the agent operated under, what decision it made, what the failure mode was, and what governance change prevents recurrence.
**Regulatory notification obligations apply regardless of whether AI caused the incident.**
**Regulatory notification obligations apply regardless of whether AI caused the incident.**
GDPR Article 33/34 and EU AI Act incident reporting obligations are not suspended because an AI system caused or contributed to the incident. The notification timeline and threshold are unchanged.
**Test incident response for AI-specific scenarios proactively.**
**Test incident response for AI-specific scenarios proactively.**
Standard chaos engineering and resilience drills must include AI-specific scenarios: prompt injection, agent scope violation, agentic hallucination triggering a downstream action. Untested playbooks do not work under pressure.
---
## 10. Deterministic Execution
**Prefer deterministic code over repeated AI inference for repeatable, well-specified tasks.**
**Prefer deterministic code over repeated AI inference for repeatable, well-specified tasks.**
If a task has a correct answer that does not depend on context or judgement, encode it as a script. Use AI once to generate and review the script; run the script in production. Repeated AI inference for a deterministic task adds cost, unreliability, and attack surface without benefit.
**Use AI inference at execution time only for tasks that are genuinely ambiguous or context-dependent.**
**Use AI inference at execution time only for tasks that are genuinely ambiguous or context-dependent.**
Applying probabilistic AI to deterministic problems is a documented anti-pattern. If you can draw a complete flowchart of the process with no "it depends" branches, the task does not need AI at execution time.
**AI-generated scripts are first drafts, not finished artefacts.**
**AI-generated scripts are first drafts, not finished artefacts.**
Review AI-generated code for correctness, missing dependencies, and performance before production deployment. EffiBench (2024) found measurable execution overhead in unreviewed AI-generated code; human review substantially closes that gap. The review step is not optional.
**Deterministic enforcement must sit outside the AI, not inside it.**
**Deterministic enforcement must sit outside the AI, not inside it.**
Linters, CI gates, unit tests, and schema validation must run on AI-generated code as hard constraints. AI instructions alone are probabilistic and cannot serve as enforcement mechanisms.
**The script is the governed artefact; version and review it accordingly.**
**The script is the governed artefact; version and review it accordingly.**
When a repeatable task changes enough to invalidate the existing script, that is the trigger to re-engage AI — not a reason to revert to repeated inference. The script lives in version control, is human-reviewable, and is the authoritative record of how the task is performed.
---
## Governance
**This document is a living artifact.**
**This document is a living artifact.**
It must be reviewed after any significant AI incident, at each major addition of AI tooling, and at minimum annually. Research that contradicts current principles must be incorporated.
**Principles without enforcement are claims.**
**Principles without enforcement are claims.**
Each principle above must map to at least one verifiable behaviour, automated check, or documented review process. Where that mapping does not exist, the principle is aspirational — label it as such and set a deadline for operationalisation. `core/instructions/governance.md` provides the agent-actionable distillation of this document; deterministic tooling (linters, CI gates, secret scanners, licence scanners) provides the enforcement layer that agent instructions alone cannot.
*Example mapping — Section 2, "Secrets must never enter AI context":*
@@ -228,11 +228,11 @@ Each principle above must map to at least one verifiable behaviour, automated ch
- CI gate: secret scanning step in pipeline rejects commits containing high-entropy strings
- Review checklist item: confirm no secrets in prompt logs before any session transcript is stored or shared
**This constitution does not replace legal advice.**
**This constitution does not replace legal advice.**
It operationalises current regulatory and research consensus for practitioners. For jurisdiction-specific obligations, regulatory filings, or IP disputes, consult qualified legal counsel.
---
*Derived from: AI Governance Research Session (May 2026).*
*Research documentation: `docs/research/governance_principles/ai-governance-research.md` | Open challenges: `docs/research/governance_principles/ai-governance-research-challenges.md`*
*Derived from: AI Governance Research Session (May 2026).*
*Research documentation: `docs/research/governance_principles/ai-governance-research.md` | Open challenges: `docs/research/governance_principles/ai-governance-research-challenges.md`*
*Operative files: `core/instructions/governance.md` (agent instructions) | `docs/HUMANS.md` (human practitioner rules) | `docs/research/governance_principles/CONTROLS.md` (deterministic enforcement)*
+1 -1
View File
@@ -1,6 +1,6 @@
# 0015 — AGENTS.md refactor (prerequisite)
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+2 -2
View File
@@ -1,6 +1,6 @@
# 0016 — Second grill: skill implementation workflow
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
@@ -29,7 +29,7 @@ HITL: requires human participation in the grill session.
## Handoff
**Status:** complete
**Status:** complete
**Files produced:**
- `docs/notes/skill-implementation-workflow.md`
+1 -1
View File
@@ -1,6 +1,6 @@
# 0017 — factory/write-eval (bootstrap skill)
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0018 — factory/write-skill (bootstrap skill)
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0019 — Factory skills: write-adr, write-issue-spec, write-workflow, upgrade-skill, validate-skill
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0020 — Design skills: grill-lean, grill-me, write-prd, architecture-review, break-into-issues, prototype
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0021 — Implement skills: implement-feature, tdd, refactor, diagnose
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0022 — Test skills: write-tests, generate-test-data, review-test-coverage
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0023 — Review skills + cliff.toml: code-review, security-review, pr-description, changelog-entry
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0024 — Deploy skills: write-ci-pipeline, write-deployment-config, write-ai-review-workflow, deployment-checklist
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0025 — Operate skills: write-runbook, incident-diagnosis, post-mortem, inspect-deployment
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0026 — IaC skills: write-docker-compose, iac-security-review
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0027 — Cross-cutting skills: session-handoff, governance-check, git-commit-message, improve-codebase-architecture, triage, zoom-out, caveman
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
+1 -1
View File
@@ -1,6 +1,6 @@
# 0028 — Chunk 3 closure: update skills-index, update spec, behavioral tests
**Type:** HITL
**Type:** HITL
**Parent PRD:** `docs/prd/chunk-3-skills-library.md`
## What to build
@@ -20,21 +20,21 @@ This changes almost every downstream decision. Until it's resolved, individual g
### 2.1 Where coding conventions live
**Factory research:** Preferred/Avoid code blocks belong in `CONTEXT.md`. Convention lives in a single shared-vocabulary file the agent always loads.
**Factory research:** Preferred/Avoid code blocks belong in `CONTEXT.md`. Convention lives in a single shared-vocabulary file the agent always loads.
**Chunk 2 plan:** Coding conventions go in a separate `core/instructions/coding.md` file, loaded on-demand.
One of these wins. If conventions go in `CONTEXT.md`, `coding.md` becomes much thinner or unnecessary. If they stay in `coding.md`, the factory's "CONTEXT.md is the primary context artefact" principle weakens.
### 2.2 Skill path structure: flat vs. nested
**Factory research:** 34 skills in 9 nested categories — `design/grill-me/SKILL.md`, `implement/tdd/SKILL.md`, `cross-cutting/session-handoff/SKILL.md`, etc. (phase × domain matrix)
**Factory research:** 34 skills in 9 nested categories — `design/grill-me/SKILL.md`, `implement/tdd/SKILL.md`, `cross-cutting/session-handoff/SKILL.md`, etc. (phase × domain matrix)
**Current state:** 12 skills in a flat structure — `grill-me/SKILL.md`, `tdd/SKILL.md`, etc.
Changing to nested paths breaks any tool that discovers skills by path. The nested structure also implies a different deployment model from `install.sh`. This is a structural decision for Chunk 3 — if we're adopting the nested taxonomy, it needs to be decided before writing any more skill files.
### 2.3 Governance file naming
**Factory research:** `AGENTS.md` at repo root as the single governance file for agents.
**Factory research:** `AGENTS.md` at repo root as the single governance file for agents.
**Current repo:** `core/instructions/governance.md` loaded via `@import` in `providers/claude-code/CLAUDE.md`.
These serve the same purpose. The factory's naming is cleaner (one file, obvious name), but the current structure fits the provider-agnostic model (governance isn't provider-specific). Not a blocking conflict, but naming inconsistency will cause confusion in grill sessions if not resolved.
+9 -9
View File
@@ -1,6 +1,6 @@
# Skill Implementation Workflow
**Produced by:** issue 0016 grill session, 2026-05-17
**Produced by:** issue 0016 grill session, 2026-05-17
**Applies to:** all Chunk 3 skill issues (0017–0028)
---
@@ -84,7 +84,7 @@ This is not a full grill-with-docs session — it is focused and bounded. If the
Work through the following in order, iterating with the human. Sub-agents handle writing tasks where context accumulation is a risk.
**a. Trigger description**
**a. Trigger description**
Write the `description:` frontmatter field first. Test it against three cases before writing the body:
1. Explicit invocation — user says the trigger phrase directly
2. Implicit invocation — user describes the task without the trigger phrase
@@ -92,7 +92,7 @@ Write the `description:` frontmatter field first. Test it against three cases be
For each case, output an explicit **PASS** or **FAIL** result. Do not proceed to step b until all three show PASS. Including the description inside the section walk-through (step b) does not satisfy this gate — it must be a standalone test-then-proceed step with per-case verdicts. If any case fails, revise the description and re-test before continuing.
**b. Per-section options walk-through**
**b. Per-section options walk-through**
Before writing anything, walk through each body section with the human. For each section:
- State what content is proposed and which upstream source it comes from
- Present alternatives where upstream sources offered different approaches
@@ -100,24 +100,24 @@ Before writing anything, walk through each body section with the human. For each
Do not write the SKILL.md until the human has confirmed every section. The synthesis grill decisions cover the eval schema and gating questions; this step covers how upstream content maps to each SKILL.md section. These are separate conversations — do not collapse them.
**c. SKILL.md** (sub-agent)
**c. SKILL.md** (sub-agent)
Once all sections are confirmed, spawn a write agent to produce the SKILL.md using `write-skill` (or hand-write for bootstrap skills). The agent receives: trigger description, per-section decisions from step b, upstream content to incorporate, authoring standard (see below).
**c. META.md — `source:` and `references:` fields**
**c. META.md — `source:` and `references:` fields**
Populate `META.md` after upstream review. Two distinct fields:
- `source:` — upstream provenance tracking (repo slug, commit SHA, files adopted with inline comments, updated date). Present only if content was adopted. Absence = self-authored.
- `references:` — general citations (research papers, documentation, standard specifications). Present only if the skill cites external research.
Both fields live in `META.md` alongside the SKILL.md — not in frontmatter. See `META-TEMPLATE.md` in `.agents/skills/write-skill/` for the full schema.
**d. eval.yaml** (sub-agent)
**d. eval.yaml** (sub-agent)
Invoke `write-eval` in two steps to preserve its confirmation gate:
1. Sub-agent proposes test cases and returns the plan to the main conversation.
2. Human confirms the plan; then sub-agent writes the file.
Do not pass pre-designed test cases directly to a write agent — that collapses the plan-then-confirm gate into a single step, bypassing write-eval's own constraint. Co-located at `.agents/evals/<category>/<skill-name>/eval.yaml`. Must contain all five required test types (see Eval schema below).
**e. HITL behavioral test**
**e. HITL behavioral test**
Human opens a fresh Claude session, invokes the skill with its trigger phrase, and verifies output. Do not batch more than 2–3 skills before running behavioral tests — output volume must stay within genuine human review capacity. An approval that cannot be meaningfully evaluated is not an approval.
### Step 6 — Session handoff
@@ -127,7 +127,7 @@ After the behavioral test passes, close the skill session by appending a `## Han
```markdown
## Handoff
**Status:** complete
**Status:** complete
**Files produced:**
- `.agents/skills/<name>/SKILL.md`
- `.agents/evals/<category>/<name>/eval.yaml`
@@ -193,7 +193,7 @@ When refactoring an existing Pocock placeholder skill:
## Eval schema
**Location:** `.agents/evals/<category>/<skill-name>/eval.yaml` — committed to the repo.
**Location:** `.agents/evals/<category>/<skill-name>/eval.yaml` — committed to the repo.
**Enforcement:** CI gates are Chunk 6. The files document expected behaviour before then.
Every eval must contain all five required test types:
+2 -2
View File
@@ -1,7 +1,7 @@
# PRD: Chunk 3 — Skills Library Rebuild
**Status:** In progress — 0015 ✅, 0016 ✅
**Produced by:** grill-with-docs session, 2026-05-17
**Status:** In progress — 0015 ✅, 0016 ✅
**Produced by:** grill-with-docs session, 2026-05-17
**Prerequisite:** AGENTS.md refactor issue must be completed before skill implementation begins
---
+3 -3
View File
@@ -1,8 +1,8 @@
# PRD: Governance Instruction Layer (Phase 1)
**Workstream:** Governance (parallel, not a numbered chunk)
**Phase:** 1 of 2 — instruction and documentation layer
**Must complete before:** Chunk 3
**Workstream:** Governance (parallel, not a numbered chunk)
**Phase:** 1 of 2 — instruction and documentation layer
**Must complete before:** Chunk 3
**Phase 2 spec:** `docs/research/governance_principles/CONTROLS.md` — deferred to Chunk 6
---
@@ -1,9 +1,9 @@
# AI Coding Factory — Implementation Guidance
**Version:** 1.2
**Date:** May 2026
**Status:** Guidance — not yet adapted to repo vision, roadmap, or ADRs
**Audience:** Human implementing the factory; Claude Code executing against it
**Version:** 1.2
**Date:** May 2026
**Status:** Guidance — not yet adapted to repo vision, roadmap, or ADRs
**Audience:** Human implementing the factory; Claude Code executing against it
**Prerequisite reading:** `ai-coding-factory-principles.md`, `ai-coding-factory-research.md`, `ai-coding-factory-skills-index.md`
> **Important:** This document is research-derived implementation guidance, not a finalised plan. It must be reconciled with the repo's actual vision and roadmap (to be established via grill-me sessions in Claude Code) before implementation begins. Where this guidance conflicts with those outputs, the grill-me session outputs win. ADRs produced in Claude Code are the authoritative implementation decisions; this document provides the evidence base and recommendations that inform them.
@@ -698,15 +698,14 @@ The intended final state: every concrete recommendation in this document either
```
/grill-me
Context: I'm building an AI coding factory — a governed, AI-assisted development environment
implemented as committed files in a repo. I have completed research and implementation
Context: I'm building an AI coding factory — a governed, AI-assisted development environment
implemented as committed files in a repo. I have completed research and implementation
guidance (loaded in context). I need to establish [session topic] before building.
Grill me until every decision is explicit. Do not let me proceed with vague answers.
Grill me until every decision is explicit. Do not let me proceed with vague answers.
Output: a structured decision record suitable for converting to an ADR.
```
---
*This is a living guidance document. Update it as grill-me sessions produce decisions. When an ADR supersedes a recommendation, mark the section with `> Superseded by ADR-NNN` and link the ADR. Version in git alongside the rest of the factory.*
+31 -31
View File
@@ -1,8 +1,8 @@
# Deterministic Controls
Applies to: any environment, repository, or pipeline where AI tools are used.
Full governance context: `docs/ai-constitution.md` — principles these controls enforce.
Human practitioner rules: `docs/HUMANS.md` | Agent instructions: `core/instructions/governance.md`
Applies to: any environment, repository, or pipeline where AI tools are used.
Full governance context: `docs/ai-constitution.md` — principles these controls enforce.
Human practitioner rules: `docs/HUMANS.md` | Agent instructions: `core/instructions/governance.md`
This file specifies the enforcement layer: controls that run mechanically, regardless of human or agent intention.
**Why this file exists:** Agent instructions and human practitioner rules are probabilistic — they depend on attention and intent. This layer removes that dependency. A control that runs automatically in CI enforces a principle more reliably than any instruction in any file. Where a principle can be enforced deterministically, it must be.
@@ -15,16 +15,16 @@ Controls configured once per development environment. Any machine or environment
---
**Secret scanning in pre-commit**
A pre-commit hook that detects secrets, credentials, API keys, and high-entropy strings must be active in every development environment. It must run before any commit reaches version control — not as a best-effort scan, but as a blocking gate.
**Secret scanning in pre-commit**
A pre-commit hook that detects secrets, credentials, API keys, and high-entropy strings must be active in every development environment. It must run before any commit reaches version control — not as a best-effort scan, but as a blocking gate.
*Enforces: Constitution §2 — secrets never enter AI context or version control.*
**AI tool data tier verification**
AI tools used for Internal, Confidential, or Restricted data must be configured to use enterprise-tier endpoints. Verify contractual data-not-trained commitments are in place before connecting any non-Public data source to an AI tool. This is a one-time verification per tool, repeated when tools or plans change.
**AI tool data tier verification**
AI tools used for Internal, Confidential, or Restricted data must be configured to use enterprise-tier endpoints. Verify contractual data-not-trained commitments are in place before connecting any non-Public data source to an AI tool. This is a one-time verification per tool, repeated when tools or plans change.
*Enforces: Constitution §3 — consumer and free-tier products handle Public data only.*
**Governance instruction file present and adapter files configured**
Every repository or project context in active use must have an agent governance instruction file present and accessible (in this repo: `core/instructions/governance.md`, deployed globally via `@import`), with tool-specific adapter files (CLAUDE.md, copilot-instructions.md, etc.) referencing it. Verify this is in place before starting AI-assisted work in any new repo.
**Governance instruction file present and adapter files configured**
Every repository or project context in active use must have an agent governance instruction file present and accessible (in this repo: `core/instructions/governance.md`, deployed globally via `@import`), with tool-specific adapter files (CLAUDE.md, copilot-instructions.md, etc.) referencing it. Verify this is in place before starting AI-assisted work in any new repo.
*Enforces: Constitution §1 — governance rules must reach the agents operating in context.*
---
@@ -35,32 +35,32 @@ Controls configured for each repository. Apply these when creating a new repo or
---
**Secret scanning in CI**
Every repository CI pipeline must include a secret scanning step that fails the build on detected credentials, tokens, or high-entropy strings. Pre-commit hooks can be bypassed; CI cannot. Both layers are required.
**Secret scanning in CI**
Every repository CI pipeline must include a secret scanning step that fails the build on detected credentials, tokens, or high-entropy strings. Pre-commit hooks can be bypassed; CI cannot. Both layers are required.
*Enforces: Constitution §2 — architectural constraint, not a reminder.*
**Dependency and security scanning**
Every repository CI pipeline must include dependency vulnerability scanning covering known CVEs and supply chain risks. For repositories using AI-generated code, the scan must be configured to cover AI-assisted contributions — not just declared dependencies.
**Dependency and security scanning**
Every repository CI pipeline must include dependency vulnerability scanning covering known CVEs and supply chain risks. For repositories using AI-generated code, the scan must be configured to cover AI-assisted contributions — not just declared dependencies.
*Enforces: Constitution §2 — AI-generated code is untrusted by default; OWASP LLM supply chain risks.*
**Licence scanning**
Every repository CI pipeline must include a licence scanning step that detects copyleft-licensed fragments (GPL, AGPL, LGPL) in committed code. Manifest-based scanners alone are insufficient for AI-assisted contributions — the scan must cover code content, not just declared dependencies.
**Licence scanning**
Every repository CI pipeline must include a licence scanning step that detects copyleft-licensed fragments (GPL, AGPL, LGPL) in committed code. Manifest-based scanners alone are insufficient for AI-assisted contributions — the scan must cover code content, not just declared dependencies.
*Enforces: Constitution §8 — copyleft fragments can appear in AI output without headers.*
**AI agent permission scoping**
Any AI agent granted access to this repository must be configured with the minimum permissions required for its specific task. Broad-scope tokens granting read/write access to the full repository or infrastructure are prohibited for AI agents. Token scope must be documented and reviewed when the agent's task scope changes.
**AI agent permission scoping**
Any AI agent granted access to this repository must be configured with the minimum permissions required for its specific task. Broad-scope tokens granting read/write access to the full repository or infrastructure are prohibited for AI agents. Token scope must be documented and reviewed when the agent's task scope changes.
*Enforces: Constitution §2 — least-privilege for all AI agents.*
**Prompt version control**
Any prompt used in an automated or recurring AI pipeline — not ad-hoc sessions — must be committed to version control with a change history. Prompts not under version control are not auditable. A prompt that runs in production without version control is uncontrolled code.
**Prompt version control**
Any prompt used in an automated or recurring AI pipeline — not ad-hoc sessions — must be committed to version control with a change history. Prompts not under version control are not auditable. A prompt that runs in production without version control is uncontrolled code.
*Enforces: Constitution §7 — prompts are code; unversioned prompts are unauditable.*
**Audit logging for agentic workflows**
Any agentic workflow that modifies state — files, infrastructure, configuration, deployments — must produce a log capturing: prompt input (or reference to versioned prompt), model version, action taken, outcome, timestamp. The log must be tamper-evident and human-readable. Isolated timestamps without action context are not sufficient.
**Audit logging for agentic workflows**
Any agentic workflow that modifies state — files, infrastructure, configuration, deployments — must produce a log capturing: prompt input (or reference to versioned prompt), model version, action taken, outcome, timestamp. The log must be tamper-evident and human-readable. Isolated timestamps without action context are not sufficient.
*Enforces: Constitution §7 — every agent action producing an effect must generate a trace.*
**Human approval gate for production changes**
Any CI/CD pipeline that applies changes to production systems, security configuration, or infrastructure must include an explicit human approval step before the change is applied. Automated merge-and-deploy pipelines for AI-generated changes are prohibited without this gate. The gate must be implemented in the pipeline, not left to individual judgment.
**Human approval gate for production changes**
Any CI/CD pipeline that applies changes to production systems, security configuration, or infrastructure must include an explicit human approval step before the change is applied. Automated merge-and-deploy pipelines for AI-generated changes are prohibited without this gate. The gate must be implemented in the pipeline, not left to individual judgment.
*Enforces: Constitution §5 — production requires a human checkpoint; this is a hard rule.*
---
@@ -71,25 +71,25 @@ Controls that must be verified periodically. These cannot be configured once and
---
**Pre-commit hook integrity** *(per developer, monthly)*
**Pre-commit hook integrity** *(per developer, monthly)*
Verify pre-commit hooks are installed, active, and current in every active development environment. Hooks can be bypassed, uninstalled by tooling updates, or silently disabled. A hook that is not tested is not a control.
**CI scan results review** *(per repository, per release or sprint)*
**CI scan results review** *(per repository, per release or sprint)*
Review secret, licence, and dependency scan outputs — not just pass/fail status. A scan that passes because exceptions have accumulated is not a clean scan. Review exception lists and remove expired or unjustified exceptions.
**AI agent permission audit** *(per repository, quarterly)*
**AI agent permission audit** *(per repository, quarterly)*
Verify that AI agent tokens and permissions remain scoped to current task requirements. Agent permissions granted for a specific task tend to persist after the task ends. Revoke and re-scope on a defined cadence.
**Audit log review** *(per agentic workflow, per sprint or monthly)*
**Audit log review** *(per agentic workflow, per sprint or monthly)*
Review AI agent action logs for unexpected scope, anomalous patterns, or actions that should have triggered a human approval gate but did not. Logging without review is record-keeping, not oversight.
**AI deployment value review** *(per deployment, time-bounded)*
**AI deployment value review** *(per deployment, time-bounded)*
Every AI integration must be reviewed against the success criteria defined before deployment. Integrations that have not delivered measurable value within the defined review period must be redesigned or discontinued. Schedule this review at deployment time, not retrospectively.
**Provider terms and data handling review** *(annually, or when providers update terms)*
**Provider terms and data handling review** *(annually, or when providers update terms)*
Verify that AI provider terms of service, data handling commitments, and IP provisions remain consistent with what was agreed at onboarding. Provider terms change. An enterprise commitment made in 2024 may not have the same scope in 2026. Re-verify; do not assume continuity.
**Constitution and controls alignment review** *(annually, or after any significant AI incident)*
**Constitution and controls alignment review** *(annually, or after any significant AI incident)*
Verify that the controls specified here remain aligned with the current version of the AI Constitution. When the constitution is updated, this file must be reviewed and updated to match. A control specification that drifts from the constitution it enforces is not a control.
---
@@ -102,5 +102,5 @@ Human judgment decisions — which AI model to use, whether a specific output is
---
*Derived from AI Constitution v1.1 — May 2026.*
*Derived from AI Constitution v1.1 — May 2026.*
*Counterpart to: `docs/HUMANS.md` | `core/instructions/governance.md` | Full context: `docs/ai-constitution.md`*
@@ -2,8 +2,8 @@
**Purpose:** Auditability of the artifact creation process. Documents what was done, how, why, and what decisions were made or deferred. Not a task list — a process record.
**Project:** AI governance research and artifact creation for a software development, deployment, and infrastructure management context.
**Sessions:** Three sessions, May 2026.
**Project:** AI governance research and artifact creation for a software development, deployment, and infrastructure management context.
**Sessions:** Three sessions, May 2026.
**Artifacts produced:** See artifact registry below.
---
@@ -28,8 +28,8 @@ The work was deliberately sequenced: research first, then distil into operative
Each topic in the research document follows: question being researched → findings → counterarguments and challenges → bias flag → provisional principles.
### Distillation logic
Research document = full sourced reasoning (human reference, never in agent context).
Constitution = concise principles derived from research (agent-readable, repo artifact).
Research document = full sourced reasoning (human reference, never in agent context).
Constitution = concise principles derived from research (agent-readable, repo artifact).
AGENTS.md = agent-actionable subset of the constitution, optimised for context window efficiency.
### Standing integrity caveat
@@ -4,7 +4,7 @@
**How to use this document:** This is the research layer, not the operative layer. If you want to know *what to do*, read `ai-constitution.md`. If you want to know *why a principle exists*, challenge a finding, or update the evidence base, read the relevant topic here. Each topic ends with provisional principles that map directly to a constitution section — cross-references are noted.
**Session:** May 2026
**Session:** May 2026
**Methodology:** Topic-by-topic web research from reliable sources. All conclusions are provisional and challengeable. Research must remain unbiased — findings drive principles, not the other way around.
---
+2 -2
View File
@@ -47,7 +47,7 @@ Current skills (direct): `caveman`, `diagnose`, `gitleaks`, `grill-me`, `grill-w
- `init-project.sh` — bootstraps a new project (Chunk 6)
- Copilot provider adapter (Chunk 7)
- Formal CI/pre-commit enforcement of governance rules (Chunk 6)
- `setup-gitleaks.sh` not yet wired into `init-project.sh` (Chunk 6) — run manually against new repos
- `init-project.sh` scaffolding (Chunk 6) — will seed `.pre-commit-config.yaml` for new projects
For chunk planning and open questions, see `docs/ROADMAP.md`.
@@ -57,7 +57,7 @@ For chunk planning and open questions, see `docs/ROADMAP.md`.
- 2026-06-20 — kyberforge plugin created and registered in `holocron` marketplace. Consolidates `create-plugin`, `marketplace-architect`, `write-skill`, and `write-eval` skills (previously in `.agents/skills/`) plus their evals, bundled scripts, references, and the plugin-marketplace-architecture research doc into a single installable plugin at `plugins/kyberforge/`. Plugin template (`templates/plugin/`) bundled into `plugins/kyberforge/skills/create-plugin/assets/plugin-template/` and removed from repo root. Evals moved from `.agents/evals/marketplace/` and `.agents/evals/factory/write-eval/` into `plugins/kyberforge/tests/evals/`. These four skills are no longer available as standalone slash commands — install the plugin to use them.
- 2026-06-20 — Gitleaks secret scanning added. `scripts/setup-gitleaks.sh` installs gitleaks v8.24.2, seeds `.gitleaks.toml` (first run only — project-owned after that), and writes a managed pre-commit hook block that is replaced on re-run. `scripts/gitleaks.toml` is the base config template extending gitleaks defaults. `tests/test-setup-gitleaks.sh` covers 6 behaviors (reject non-git dir, config deploy, hook create, append, stale-block replace, idempotency). `.gitleaks.toml` in repo root adds path allowlist for `docs/research/` (high-entropy terminal captures). `gitleaks` skill added (`cross-cutting`) covering full lifecycle: install, update, tune allowlist, scan modes, resolve real findings. Key lesson: v8.24.2 uses `[allowlist]` (singular); v8.25.0+ uses `[[allowlists]]` — wrong syntax silently does nothing.
- 2026-06-20 — Gitleaks secret scanning added via pre-commit framework. `.gitleaks.toml` in repo root is the base config extending gitleaks defaults and adds path allowlist for `docs/research/` (high-entropy terminal captures). `gitleaks` skill added (`cross-cutting`) covering full lifecycle: install, update, tune allowlist, scan modes, resolve real findings. Supports both modern repos (pre-commit-based) and legacy repos (shell hook-based setup). Key lesson: v8.24.2 uses `[allowlist]` (singular); v8.25.0+ uses `[[allowlists]]` — wrong syntax silently does nothing.
- 2026-05-18 — Issue 0018 phase 1 refactor complete: `write-skill` redesigned from scratch. New files added to skill directory: `SKILL-TEMPLATE.md` (authoritative 6-section template with XML blocks, human-usable), `META-TEMPLATE.md` (provenance schema with inline-commented YAML), `CATEGORIES.md` (self-contained category table), `META.md` (write-skill's own provenance). SKILL.md rewritten: 6 sections replacing 8 (Role and When/When not dropped — not in agentskills.io spec); frontmatter reduced to 3 fields (`name`, `description`, `metadata.category`); provenance fields (`version`, `updated`, `when`, `source`, `references`) moved to META.md (progressive disclosure — not loaded at startup). `docs/notes/skill-implementation-workflow.md` updated to reference SKILL-TEMPLATE.md as the authoritative template.
@@ -1,84 +0,0 @@
skill_name: gitleaks
trigger_tests:
- id: explicit-trigger-install
name: Explicit trigger — install and configure
query: "set up gitleaks in this repo"
should_trigger: true
- id: explicit-trigger-update-hook
name: Explicit trigger — update hook
query: "update the gitleaks pre-commit hook"
should_trigger: true
- id: implicit-trigger-false-positive
name: Implicit trigger — suppress false positive in pre-commit hook
query: "my pre-commit hook keeps blocking commits because it thinks my test fixture has an API key, how do I suppress it?"
should_trigger: true
- id: implicit-trigger-scan-history
name: Implicit trigger — audit repo history for secrets
query: "I want to scan my entire git history to make sure no credentials were ever committed"
should_trigger: true
- id: negative-trigger-security-review
name: Negative trigger — general code security review
query: "do a security review of this pull request"
should_trigger: false
- id: negative-trigger-ci-scan
name: Negative trigger — CI pipeline secret scanning
query: "add secret scanning to my GitHub Actions CI pipeline"
should_trigger: false
- id: negative-trigger-other-tool
name: Negative trigger — different secret scanning tool
query: "trufflehog is flagging false positives in my repo, how do I configure it?"
should_trigger: false
output_tests:
- id: output-states-intent-before-running
name: Deterministic — states intent and waits for confirmation before running setup script
type: deterministic
prompt: "set up gitleaks in /home/user/myproject"
expected_output: >
The skill states what it is about to do (run setup-gitleaks.sh, which will modify
.git/hooks/pre-commit and seed .gitleaks.toml) and asks for confirmation before
executing the script. The confirmation request appears before any Bash command
that runs the script.
assertions:
- "Output contains 'setup-gitleaks.sh' before any Bash tool invocation that runs it"
- "Output contains a confirmation prompt (e.g. 'Proceed?') before running the setup script"
- "Output does not run the setup script without an explicit confirmation step"
- id: output-redacts-secret-values
name: Deterministic — redacts secret value and prescribes rotation before history rewrite
type: deterministic
prompt: "gitleaks found an AWS secret key on line 42 of config/settings.py — it says the value is AKIAIOSFODNN7EXAMPLE. What do I do?"
expected_output: >
The skill does not echo or repeat the secret value. It references the finding by
rule ID, file path, and line number only. It states that the secret must be rotated
immediately. It mentions git filter-repo (not git filter-branch) for history rewrite
and asks for user confirmation before running the rewrite.
assertions:
- "Output does not contain the literal string 'AKIAIOSFODNN7EXAMPLE'"
- "Output mentions rotating or revoking the credential as the first action"
- "Output references 'git filter-repo' for history rewrite, not 'git filter-branch'"
- "Output states that user confirmation is required before running the history rewrite"
- id: output-quality-allowlist-guidance
name: LLM-rubric — allowlist guidance is correct, version-aware, and minimal
type: llm-rubric
prompt: "gitleaks keeps flagging my docs/research/ directory as containing secrets, how do I suppress it?"
expected_output: >
High-quality output checks the installed gitleaks version before prescribing any
TOML syntax, recommends a path-based allowlist entry in .gitleaks.toml (not a
.gitleaksignore fingerprint), uses the correct TOML syntax for the detected version,
adds only the minimal allowlist entry needed for the identified false positive, and
includes a verification step (re-run gitleaks dir -v or gitleaks dir --log-level debug)
after making the change.
assertions:
- "Output checks or asks about the gitleaks version before writing TOML syntax"
- "Output recommends a path-based allowlist entry in .gitleaks.toml rather than .gitleaksignore"
- "Output includes a command to verify the suppression works after the change"
- "Output explains why .gitleaksignore fingerprints are fragile (line numbers shift)"
@@ -1,97 +0,0 @@
skill_name: neuledge-context
trigger_tests:
- id: explicit-install-register
name: "Explicit trigger — install and register"
query: "install @neuledge/context and register it as an MCP server in Claude Code"
should_trigger: true
- id: explicit-package-management
name: "Explicit trigger — package management"
query: "install the react documentation package using neuledge context"
should_trigger: true
- id: implicit-offline-docs
name: "Implicit trigger — offline docs for AI agent"
query: "I need React and Next.js docs available to my AI agent without web searches"
should_trigger: true
- id: negative-different-mcp
name: "Negative — different MCP server"
query: "Add the Gitea MCP server to Claude Code"
should_trigger: false
- id: negative-query-existing
name: "Negative — querying an already-running server"
query: "How do I query React docs using the context server that's already running?"
should_trigger: false
- id: negative-cursor-setup
name: "Negative — different provider"
query: "Set up context serve for Cursor"
should_trigger: false
output_tests:
- id: install-script-announced
name: "Install — script announced before running, version verified after"
type: deterministic
prompt: "install @neuledge/context"
expected_output: >
The skill announces that it will run scripts/setup-neuledge-context.sh before executing it,
then verifies the installation by running `context --version`.
assertions:
- "Output mentions 'scripts/setup-neuledge-context.sh' before any install command is run"
- "Output includes a `context --version` call after the install step"
- "Output does not contain `npm install -g @neuledge/context@latest` (no floating @latest)"
- id: mcp-list-before-add
name: "MCP registration — list checked before add, skipped if present"
type: deterministic
prompt: "register @neuledge/context as a Claude Code MCP server"
expected_output: >
The skill runs `claude mcp list` and checks for an existing 'context' entry before
running `claude mcp add`. If already registered, the add step is skipped.
assertions:
- "Output includes `claude mcp list` before `claude mcp add context`"
- "Output states that registration is skipped when the server is already present"
- "The `claude mcp add` command uses stdio form: `claude mcp add context -- context serve`"
- id: auth-chmod-paired
name: "Auth — secure-context-config.sh run immediately after auth add"
type: deterministic
prompt: "add auth credentials for docs.example.com to neuledge context"
expected_output: >
The skill runs `context auth add docs.example.com` with an environment variable reference
for the credential, then immediately runs scripts/secure-context-config.sh.
No credential value appears in the output.
assertions:
- "Output references an environment variable (e.g. $TOKEN) rather than a literal credential value"
- "Output runs `scripts/secure-context-config.sh` in the same step as or immediately after `context auth add`"
- "No bearer token, cookie value, or other credential string appears in the output"
- id: git-check-url-source
name: "context add — git prerequisite checked for URL sources only"
type: deterministic
prompt: "add documentation from https://github.com/prisma/prisma using context add"
expected_output: >
Before running `context add`, the skill checks `git --version` because the source is a
GitHub URL. The check is present for URL/repo sources and absent for local .db file paths.
assertions:
- "Output includes `git --version` before the `context add https://github.com/...` command"
- "If given a local .db file path instead, the git check is absent"
- id: install-flow-quality
name: "Full install + register flow quality"
type: llm-rubric
prompt: "install @neuledge/context and set it up as my Claude Code MCP server"
expected_output: >
A complete, ordered install-then-register flow: (1) announce the install script,
(2) run setup-neuledge-context.sh, (3) verify with context --version,
(4) check claude mcp list, (5) run claude mcp add if not already registered,
(6) confirm with claude mcp list. Steps are in the correct order with verification
between install and registration.
assertions:
- "Install step comes before MCP registration step"
- "A verification command (context --version) appears between install and registration"
- "The output would leave a user with a working @neuledge/context MCP server in Claude Code"
- "No step is skipped without an explanation of why it was skipped"
@@ -1,61 +0,0 @@
skill_name: write-docs
trigger_tests:
- id: explicit-trigger-document-module
name: "Explicit trigger — document a script"
query: "Write documentation for the install.sh script"
should_trigger: true
- id: explicit-trigger-create-docs
name: "Explicit trigger — create docs for a feature"
query: "Create docs for this feature"
should_trigger: true
- id: implicit-trigger-readme-update
name: "Implicit trigger — outdated README section, no trigger phrase"
query: "We need to update the README section for the auth module, the current one is outdated"
should_trigger: true
- id: negative-trigger-prd
name: "Negative — PRD request should route to to-prd"
query: "Write a PRD for the new logging feature"
should_trigger: false
- id: negative-trigger-write-skill
name: "Negative — skill authoring request should route to write-skill"
query: "Write a skill for generating documentation automatically"
should_trigger: false
- id: negative-trigger-skill-file
name: "Negative — SKILL.md update (skill files are self-describing)"
query: "Document how the write-docs skill works by updating its SKILL.md"
should_trigger: false
output_tests:
- id: output-proposes-files-before-reading
name: "Deterministic — candidates proposed or approval sought before reading files"
type: deterministic
prompt: "Write documentation for the config module"
expected_output: "Skill proposes candidate files or asks the user to name specific files before reading any file content"
assertions:
- "Response proposes candidate file paths or asks the user to confirm which files to read before showing any extracted content"
- "Response does not display extracted code content or API surface without first receiving file approval"
- id: output-gap-check-present
name: "Deterministic — gap check step present before drafting"
type: deterministic
prompt: "Write documentation for the install.sh script, audience: developer"
expected_output: "Skill presents extracted behaviour to the user and asks them to fill gaps before drafting any section"
assertions:
- "Response includes a gap check step that presents extracted behaviour and asks what the code does not explain"
- "Response does not skip directly to a drafted documentation section without presenting extracted content first"
- id: output-never-invents-behaviour
name: "LLM rubric — no invented behaviour, all claims sourced"
type: llm-rubric
prompt: "Document the src/config.py file for internal developers"
expected_output: "Documentation where every claim is attributed to code content or explicit user input, with no invented explanations, assumptions about intent, or unverifiable behaviour claims."
assertions:
- "The skill explicitly derives each documented claim from a named source — a code line, spec section, or user statement — and does not add claims without attribution"
- "The skill does not include descriptions of caller intent, design rationale, or future behaviour that are not present in the source material"
- "If a behaviour is undocumentable (internal detail with no public spec), the skill notes it as out-of-scope rather than inventing an explanation"
-15
View File
@@ -1,15 +0,0 @@
```yaml
version: "1.0"
updated: 2026-06-20
when: >
Invoked when the user wants to install gitleaks and wire it as a git pre-commit secret
scanner, update the hook in an existing repo, tune allowlist rules to suppress false
positives, debug a scan finding, or rotate a real secret that was found. Covers the full
lifecycle: install → configure → maintain → remediate. Not invoked for general code
security review (security-review skill) or CI pipeline secret scanning (write-ci-pipeline skill).
references:
- https://github.com/gitleaks/gitleaks/releases/tag/v8.24.2
- https://github.com/gitleaks/gitleaks/blob/main/README.md
```
-111
View File
@@ -1,111 +0,0 @@
---
name: gitleaks
description: Use when the user wants to install gitleaks, wire it as a git pre-commit secret scanner, update the hook in an existing repo, tune allowlist rules, resolve false positives, or debug a gitleaks scan finding. Do NOT use when the user wants a general security review of code (use security-review), wants to add secret scanning to a CI pipeline (use write-ci-pipeline), or is asking about a different secret scanning tool such as trufflehog or git-secrets.
metadata:
category: cross-cutting
allowed-tools:
- Bash
- Read
- Edit
---
<requirements>
## Required inputs
- **Target repo path** — absolute path to the git repository to configure; inferred from current working directory if not stated, ask if ambiguous
- **Task type** — install/configure, update hook, tune allowlist, debug finding; inferred from the user's request
## Constraints
- Always state what you are about to do before running `setup-gitleaks.sh` — the script modifies `.git/hooks/pre-commit` and seeds `.gitleaks.toml`
- Never modify `.gitleaks.toml` if the user has not asked for allowlist changes — it is project-owned once seeded; treat it as user-controlled config
- Never run `gitleaks git` or `gitleaks dir` across the full history without warning the user it may be slow on large repos
- Redact any secret values that appear in gitleaks output before showing them to the user — show the rule ID, file, and line number only
- When the installed gitleaks version is unknown, check it with `gitleaks version` before suggesting config syntax — v8.24.2 uses `[allowlist]`; v8.25.0+ uses `[[allowlists]]`
- False positive suppression: prefer path-based allowlists in `.gitleaks.toml` over fingerprint-based entries in `.gitleaksignore` — fingerprints are line-number-sensitive and break on file edits
</requirements>
<steps>
## Process
### Install and configure
1. **Confirm target.** State: "I will run `scripts/setup-gitleaks.sh <path>` which will install gitleaks (if absent), seed `.gitleaks.toml` (first run only), and write the pre-commit hook. Proceed?" Wait for confirmation — this modifies the repo's git hook.
2. **Run setup script.** Execute from the ai-development repo root:
```
bash scripts/setup-gitleaks.sh <TARGET_REPO>
```
The script is idempotent — it replaces the gitleaks block in the hook on every run without disturbing other hook content.
3. **Verify installation.** Run `gitleaks version` to confirm the binary is available. Run `gitleaks git --staged --redact -v` in the target repo to confirm the hook would work on a staged commit (add a dummy change if needed to test).
4. **Commit `.gitleaks.toml`.** Remind the user that `.gitleaks.toml` belongs in version control so all contributors share the same allowlist rules.
### Update hook
Re-run `bash scripts/setup-gitleaks.sh <TARGET_REPO>` from the ai-development repo root. The managed block (delimited by `# managed by setup-gitleaks.sh` / `# end gitleaks` markers) is always replaced with the current version. Non-gitleaks hook content is preserved.
### Tune allowlist / resolve false positives
1. **Identify the false positive.** Run `gitleaks dir --log-level debug <path>` to see which rule fired and which allowlist entries (if any) are already active.
2. **Check the gitleaks version.** Run `gitleaks version`. Use `[allowlist]` syntax for v8.24.2; use `[[allowlists]]` syntax for v8.25.0+. Using the wrong syntax silently produces no errors but the allowlist does nothing — this is the most common configuration trap.
3. **Choose suppression strategy.** Read `.gitleaks.toml` first. See `references/allowlist-patterns.md` for syntax examples and when to use each approach:
- Path regex in `[allowlist]` — for files that can never contain real secrets (research notes, terminal captures, test fixtures). Preferred.
- Stopwords in `[allowlist]` — for placeholder patterns like "example", "changeme".
- `disabledRules` in `[extend]` — to disable a noisy default rule entirely. Use only when the rule has no value for this repo.
- `.gitleaksignore` fingerprint — last resort; breaks when the file is edited because line numbers shift.
4. **Edit `.gitleaks.toml`.** Add the minimal allowlist entry needed. Do not suppress more than the identified false positive.
5. **Verify.** Re-run `gitleaks dir -v <path>` or `gitleaks git -v` to confirm the false positive is suppressed and no real findings are hidden.
### Scan modes
| Mode | Command | When to use |
|---|---|---|
| Staged changes (pre-commit) | `gitleaks git --staged --redact -v` | What the hook runs |
| Full commit history | `gitleaks git -v` | Audit existing repo history |
| Working directory files | `gitleaks dir -v <path>` | Scan uncommitted files |
| Debug allowlists | `gitleaks dir --log-level debug <path>` | See which files are skipped and which allowlists fire |
### Resolve a real finding
1. Do not redact or show the secret value. Reference the rule ID, file, and line number only.
2. The secret is compromised the moment it was committed — rotate it immediately, regardless of whether the commit is reachable from the public remote.
3. Remove the secret from history using `git filter-repo` (not `git filter-branch`). This is a history-rewrite — confirm with the user before running. Force-push to all remotes after rewriting.
4. Add the file path to the `.gitleaks.toml` allowlist only if the file is known to be a false-positive source going forward (e.g. a test fixture). Do not add an allowlist entry to suppress a real finding that has been removed.
## Output format
No structured output file. The skill produces:
- Modified `.git/hooks/pre-commit` in the target repo (via the setup script)
- Modified `.gitleaks.toml` in the target repo (allowlist changes only, when requested)
- Terminal confirmation of what was changed and what to do next
</steps>
<checks>
## Failure handling
- `setup-gitleaks.sh` not found — stop; instruct the user to run from the ai-development repo root at `/root/ai-development/`
- Target path is not a git repository — report the error from the script and ask the user to confirm the correct path
- `gitleaks` binary not installed and download fails — report the curl/network error; direct the user to manual install at `https://github.com/gitleaks/gitleaks/releases`
- Wrong TOML syntax for installed version — detect via `gitleaks version`, show the correct syntax for that version, do not guess
## Self-check
- [ ] Target repo confirmed before running the setup script
- [ ] `gitleaks version` checked before writing any `.gitleaks.toml` allowlist syntax
- [ ] Secret values in scan output redacted before displaying to the user
- [ ] `.gitleaks.toml` edits are minimal — only the identified false positive suppressed
- [ ] After any allowlist change: re-ran scan to verify suppression works and no real findings are hidden
- [ ] For real findings: rotation step stated before history rewrite, user confirmed history rewrite before running `git filter-repo`
</checks>
@@ -1,83 +0,0 @@
# Gitleaks allowlist patterns
## Version syntax
| Version | Allowlist syntax |
|---|---|
| v8.24.2 and earlier | `[allowlist]` (singular table) |
| v8.25.0 and later | `[[allowlists]]` (array of tables) |
**Critical**: using the wrong syntax produces no error but the allowlist silently does nothing. Always check `gitleaks version` first.
## v8.24.2 syntax (this repo uses 8.24.2)
### Suppress by path regex
Use for files that can never contain real secrets (research notes, terminal captures, test fixtures, generated docs).
```toml
[allowlist]
description = "research notes and terminal captures"
paths = [
'''docs/research/.*''',
'''tests/fixtures/.*''',
]
```
### Suppress by stopword
Use for placeholder values that match secret patterns but are clearly not real.
```toml
[allowlist]
description = "placeholder values"
stopwords = ["example", "placeholder", "changeme", "your-api-key-here"]
```
### Disable a default rule entirely
Use only when a rule has no value for this repo and produces pervasive false positives.
```toml
[extend]
useDefault = true
disabledRules = ["generic-api-key"]
```
## v8.25.0+ syntax (for reference)
```toml
[[allowlists]]
description = "research notes"
paths = ['''docs/research/.*''']
[[allowlists]]
description = "placeholder values"
stopwords = ["example", "placeholder"]
```
## .gitleaksignore (fingerprint-based — last resort)
```
# Format: <fingerprint>:<line-number>
# Generated by: gitleaks git -v --report-format json | jq -r '.[] | "\(.Fingerprint):\(.StartLine)"'
abc123def456:42
```
Avoid this approach: fingerprints embed line numbers. Any edit to the file shifts line numbers and invalidates the entry, re-surfacing the false positive.
## Verification after any change
```bash
# Scan current files
gitleaks dir -v .
# Scan with debug output to see which allowlists fired
gitleaks dir --log-level debug .
# Scan commit history
gitleaks git -v
# Scan only staged changes (what the pre-commit hook runs)
gitleaks git --staged --redact -v
```
@@ -199,4 +199,4 @@
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
limitations under the License.
@@ -137,7 +137,7 @@ Here are some tips that we've found to work well in writing these descriptions:
- The description competes with other skills for Claude's attention — make it distinctive and immediately recognizable.
- If you're getting lots of failures after repeated attempts, change things up. Try different sentence structures or wordings.
I'd encourage you to be creative and mix up the style in different iterations since you'll have multiple opportunities to try different approaches and we'll just grab the highest-scoring one at the end.
I'd encourage you to be creative and mix up the style in different iterations since you'll have multiple opportunities to try different approaches and we'll just grab the highest-scoring one at the end.
Please respond with only the new description text in <new_description> tags, nothing else."""
@@ -97,7 +97,7 @@ if __name__ == "__main__":
if len(sys.argv) != 2:
print("Usage: python quick_validate.py <skill_directory>")
sys.exit(1)
valid, message = validate_skill(sys.argv[1])
print(message)
sys.exit(0 if valid else 1)
sys.exit(0 if valid else 1)
@@ -1 +1 @@
I created this myself :)
I created this myself :)
@@ -178,4 +178,4 @@ An instruction that changes nothing because the model already does it by default
A leading word is a *technique*; No-Op is a *verdict* on a line — and they cross. A leading word too weak to beat the default is a no-op (_be thorough_ when the agent is already thorough-ish), and the fix is a stronger word that passes the verdict (_relentless_), not a different technique. So the No-Op test — does it change behaviour versus the default? — is also how you grade whether a leading word is earning its repetitions. This is model-relative, not reader-relative: two people disagreeing over whether a line is a no-op disagree about the default, and settle it by running the skill, not by debate.
_Avoid_: redundant instruction, restating the obvious, belaboring
_Avoid_: redundant instruction, restating the obvious, belaboring
@@ -79,4 +79,4 @@ Use these to diagnose issues the user may be having with the skill.
- **Duplication** — the same meaning in more than one place. Costs maintenance and tokens, and inflates a meaning's prominence on the ladder past its real rank.
- **Sediment** — stale layers that settle because adding feels safe and removing feels risky. The default fate of any skill without a pruning discipline.
- **Sprawl** — a skill simply too long, even when every line is live and unique. Hurts readability and maintainability and wastes tokens. The cure is the ladder: disclose **reference** behind pointers, and split by **branch** or sequence so each path carries only what it needs.
- **No-op** — a line the model already obeys by default, so you pay load to say nothing. The test: does it change behaviour versus the default? A weak leading word (_be thorough_ when the agent is already thorough-ish) is a no-op; the fix is a stronger word (_relentless_), not a different technique.
- **No-op** — a line the model already obeys by default, so you pay load to say nothing. The test: does it change behaviour versus the default? A weak leading word (_be thorough_ when the agent is already thorough-ish) is a no-op; the fix is a stronger word (_relentless_), not a different technique.
@@ -169,4 +169,4 @@ digraph STYLE_GUIDE {
bad_2 -> bad_3;
bad_3 -> bad_4;
}
}
}
@@ -27,7 +27,6 @@ fi
COLOR_RESET=$'\033[0m'
COLOR_BOLD_BLUE=$'\033[1;34m' # dir — identity, always stable
COLOR_BLUE=$'\033[0;34m' # tokens — informational, no urgency
COLOR_CYAN=$'\033[0;36m' # model — configuration, always stable
COLOR_GREEN=$'\033[0;32m' # healthy / within limits
COLOR_AMBER=$'\033[0;33m' # attention / approaching a limit
COLOR_RED=$'\033[0;31m' # urgent / act now
+1
View File
@@ -3,6 +3,7 @@
# All paths: src relative to REPO_ROOT, dest relative to HOME.
# Format: "src:dest"
# shellcheck disable=SC2034
# Individual files — copied verbatim
DEPLOY_FILES=(
"providers/claude-code/CLAUDE.md:.claude/CLAUDE.md"
-24
View File
@@ -1,24 +0,0 @@
title = "gitleaks config"
[extend]
# Extends the default ruleset built into gitleaks.
# Remove useDefault and define [[rules]] from scratch if you want full control.
useDefault = true
# Rules to disable from the default set — uncomment and add IDs for known false positives.
# Run `gitleaks git -v` on your repo first to discover which rules fire.
# disabledRules = ["generic-api-key"]
# Global allowlist — applies to all rules.
# Note: uses [allowlist] (v8 syntax). v8.25.0+ uses [[allowlists]] (array of tables).
# Add path regexes or stopwords to suppress known false positives.
[allowlist]
description = "Known false positives — prose patterns and research session notes"
# docs/research/: high-entropy text from terminal captures in session notes
# docs/ROADMAP.md: documents known false positives, triggering the same rules
# ai-coding-factory-session.md:90 specifically: 'Token routing: Haiku/Sonnet/Opus'
paths = [
'''docs/research/.*''',
'''docs/ROADMAP\.md''',
]
-16
View File
@@ -1,16 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
# Sets secure file permissions on ~/.context/config.json.
# Run after every `context auth add` to protect stored credentials.
# Safe to run when the file does not exist yet.
CONFIG="${HOME}/.context/config.json"
if [ ! -f "$CONFIG" ]; then
echo "Skipped: ${CONFIG} does not exist — no permissions to set."
exit 0
fi
chmod 600 "$CONFIG"
echo "Secured: ${CONFIG} set to 600 (owner read/write only)."
-124
View File
@@ -1,124 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
# Sets up gitleaks as a git pre-commit hook in a target repository.
# Usage: setup-gitleaks.sh [TARGET_REPO]
# TARGET_REPO — path to the git repo to configure (default: current directory)
# Idempotent: safe to re-run; always replaces the hook block with the current version.
GITLEAKS_VERSION="8.24.2"
GITLEAKS_INSTALL_DIR="${GITLEAKS_INSTALL_DIR:-/usr/local/bin}"
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
TARGET="${1:-$(pwd)}"
HOOK_FILE="$TARGET/.git/hooks/pre-commit"
CONFIG_SRC="$SCRIPT_DIR/gitleaks.toml"
CONFIG_DEST="$TARGET/.gitleaks.toml"
MARKER="# managed by setup-gitleaks.sh"
END_MARKER="# end gitleaks"
# --- Install gitleaks if not present ---
install_gitleaks() {
local os arch tarball url tmp_dir
case "$(uname -s)" in
Linux) os="linux" ;;
Darwin) os="darwin" ;;
*)
echo "Error: unsupported OS '$(uname -s)' — install gitleaks manually from https://github.com/gitleaks/gitleaks/releases" >&2
exit 1
;;
esac
case "$(uname -m)" in
x86_64) arch="x64" ;;
aarch64 | arm64) arch="arm64" ;;
*)
echo "Error: unsupported architecture '$(uname -m)' — install gitleaks manually from https://github.com/gitleaks/gitleaks/releases" >&2
exit 1
;;
esac
tarball="gitleaks_${GITLEAKS_VERSION}_${os}_${arch}.tar.gz"
url="https://github.com/gitleaks/gitleaks/releases/download/v${GITLEAKS_VERSION}/${tarball}"
tmp_dir="$(mktemp -d)"
trap 'rm -rf "$tmp_dir"' RETURN
echo "Installing gitleaks v${GITLEAKS_VERSION}..."
curl -fsSL "$url" -o "$tmp_dir/$tarball"
tar -xzf "$tmp_dir/$tarball" -C "$tmp_dir" gitleaks
install -m 755 "$tmp_dir/gitleaks" "$GITLEAKS_INSTALL_DIR/gitleaks"
echo "Installed: $GITLEAKS_INSTALL_DIR/gitleaks"
}
if ! command -v gitleaks &>/dev/null; then
install_gitleaks
fi
# --- Validate ---
if [ ! -d "$TARGET/.git" ]; then
echo "Error: $TARGET is not a git repository" >&2
exit 1
fi
if [ ! -f "$CONFIG_SRC" ]; then
echo "Error: config template not found at $CONFIG_SRC" >&2
exit 1
fi
# --- Deploy config ---
if [ -f "$CONFIG_DEST" ]; then
echo "Skipped: $CONFIG_DEST already exists — edit it directly to customise rules."
else
cp "$CONFIG_SRC" "$CONFIG_DEST"
echo "Wrote: $CONFIG_DEST"
echo " Commit this file — it belongs in version control."
fi
# --- Deploy hook ---
hook_block() {
cat <<BLOCK
$MARKER
if command -v gitleaks &>/dev/null; then
gitleaks git --staged --redact -v
else
echo "Warning: gitleaks not installed — secret scan skipped (https://github.com/gitleaks/gitleaks/releases)" >&2
fi
$END_MARKER
BLOCK
}
write_hook() {
local hook_file="$1"
if grep -qF "$MARKER" "$hook_file"; then
# Remove old block (start marker through end marker inclusive) then append current version
awk -v start="$MARKER" -v end="$END_MARKER" '
$0 == start { skip=1; next }
skip && $0 == end { skip=0; next }
!skip { print }
' "$hook_file" > "${hook_file}.tmp" && mv "${hook_file}.tmp" "$hook_file"
hook_block >> "$hook_file"
echo "Updated: $hook_file (gitleaks block replaced)"
else
hook_block >> "$hook_file"
echo "Updated: $hook_file (gitleaks block appended to existing hook)"
fi
}
if [ -f "$HOOK_FILE" ]; then
write_hook "$HOOK_FILE"
else
{ echo '#!/usr/bin/env bash'; echo 'set -euo pipefail'; hook_block; } > "$HOOK_FILE"
chmod +x "$HOOK_FILE"
echo "Created: $HOOK_FILE"
fi
echo ""
echo "Done. Staged secrets will be scanned on every commit in $TARGET."
echo "To skip on a single commit: SKIP=gitleaks git commit ..."
-212
View File
@@ -1,212 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
# Installs git hook blocks for validation into a target repository.
# Usage: setup-hooks.sh [TARGET_REPO]
# TARGET_REPO — path to the git repo to configure (default: current directory)
# Idempotent: safe to re-run; always replaces each managed block with the current version.
SHELLCHECK_VERSION="0.10.0"
JQ_VERSION="1.7.1"
YQ_VERSION="4.44.3"
TOOL_INSTALL_DIR="${TOOL_INSTALL_DIR:-/usr/local/bin}"
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
TARGET="${1:-$(pwd)}"
MARKER="# managed by setup-hooks.sh"
END_MARKER="# end setup-hooks"
if [[ ! -d "$TARGET/.git" ]]; then
echo "Error: $TARGET is not a git repository" >&2
exit 1
fi
# --- Tool installation ---
_os() {
case "$(uname -s)" in
Linux) echo "linux" ;;
Darwin) echo "darwin" ;;
*) echo "Error: unsupported OS '$(uname -s)'" >&2; exit 1 ;;
esac
}
_arch() {
case "$(uname -m)" in
x86_64) echo "x86_64" ;;
aarch64 | arm64) echo "aarch64" ;;
*) echo "Error: unsupported architecture '$(uname -m)'" >&2; exit 1 ;;
esac
}
install_shellcheck() (
os="$(_os)"
arch="$(_arch)"
tarball="shellcheck-v${SHELLCHECK_VERSION}.${os}.${arch}.tar.xz"
url="https://github.com/koalaman/shellcheck/releases/download/v${SHELLCHECK_VERSION}/${tarball}"
tmp_dir="$(mktemp -d)"
trap 'rm -rf "$tmp_dir"' EXIT
echo "Installing shellcheck v${SHELLCHECK_VERSION}..."
curl -fsSL "$url" -o "$tmp_dir/$tarball"
tar -xJf "$tmp_dir/$tarball" -C "$tmp_dir" --strip-components=1
install -m 755 "$tmp_dir/shellcheck" "$TOOL_INSTALL_DIR/shellcheck"
echo "Installed: $TOOL_INSTALL_DIR/shellcheck"
)
install_jq() (
os="$(_os)"
[[ "$os" == "darwin" ]] && os="macos"
arch="$(_arch | sed 's/x86_64/amd64/; s/aarch64/arm64/')"
url="https://github.com/jqlang/jq/releases/download/jq-${JQ_VERSION}/jq-${os}-${arch}"
tmp_dir="$(mktemp -d)"
trap 'rm -rf "$tmp_dir"' EXIT
echo "Installing jq v${JQ_VERSION}..."
curl -fsSL "$url" -o "$tmp_dir/jq"
install -m 755 "$tmp_dir/jq" "$TOOL_INSTALL_DIR/jq"
echo "Installed: $TOOL_INSTALL_DIR/jq"
)
install_yq() (
os="$(_os)"
arch="$(_arch | sed 's/x86_64/amd64/; s/aarch64/arm64/')"
url="https://github.com/mikefarah/yq/releases/download/v${YQ_VERSION}/yq_${os}_${arch}"
tmp_dir="$(mktemp -d)"
trap 'rm -rf "$tmp_dir"' EXIT
echo "Installing yq v${YQ_VERSION}..."
curl -fsSL "$url" -o "$tmp_dir/yq"
install -m 755 "$tmp_dir/yq" "$TOOL_INSTALL_DIR/yq"
echo "Installed: $TOOL_INSTALL_DIR/yq"
)
ensure_tool() {
local tool="$1"
if ! command -v "$tool" &>/dev/null; then
"install_${tool}"
fi
}
ensure_tool shellcheck
ensure_tool jq
ensure_tool yq
# --- Marker-block helpers ---
write_block() {
local hook_file="$1"
local block_content="$2"
if grep -qF "$MARKER" "$hook_file"; then
awk -v start="$MARKER" -v end="$END_MARKER" '
$0 == start { skip=1; next }
skip && $0 == end { skip=0; next }
!skip { print }
' "$hook_file" > "${hook_file}.tmp" && mv "${hook_file}.tmp" "$hook_file"
fi
printf '\n%s\n%s\n%s\n' "$MARKER" "$block_content" "$END_MARKER" >> "$hook_file"
chmod +x "$hook_file"
}
ensure_hook() {
local hook_file="$1"
if [[ ! -f "$hook_file" ]]; then
printf '#!/usr/bin/env bash\nset -euo pipefail\n' > "$hook_file"
chmod +x "$hook_file"
fi
}
# --- commit-msg: conventional commits ---
COMMIT_MSG_HOOK="$TARGET/.git/hooks/commit-msg"
ensure_hook "$COMMIT_MSG_HOOK"
commit_msg_block() {
cat <<'BLOCK'
msg=$(cat "$1")
pattern='^(feat|fix|docs|chore|refactor|test|perf|ci|build|revert)(\(.+\))?!?: .+'
if ! echo "$msg" | grep -qE "$pattern"; then
echo "ERROR: Commit message must follow Conventional Commits format." >&2
echo " Examples: feat: add login, fix(auth): correct token expiry, chore!: drop python dep" >&2
exit 1
fi
BLOCK
}
write_block "$COMMIT_MSG_HOOK" "$(commit_msg_block)"
echo "Updated: $COMMIT_MSG_HOOK (conventional commits check)"
# --- pre-commit: shellcheck + jq/yq + SKILL.md frontmatter ---
PRE_COMMIT_HOOK="$TARGET/.git/hooks/pre-commit"
ensure_hook "$PRE_COMMIT_HOOK"
pre_commit_block() {
cat <<'BLOCK'
staged=$(git diff --cached --name-only --diff-filter=ACM)
# shellcheck on staged .sh files
if command -v shellcheck &>/dev/null; then
while IFS= read -r f; do
[[ -f "$f" ]] && shellcheck -x "$f"
done < <(echo "$staged" | grep '\.sh$' || true)
else
echo "Warning: shellcheck not installed — shell script linting skipped" >&2
fi
# jq validation on staged .json files
if command -v jq &>/dev/null; then
while IFS= read -r f; do
[[ -f "$f" ]] && jq . "$f" > /dev/null
done < <(echo "$staged" | grep '\.json$' || true)
else
echo "Warning: jq not installed — JSON validation skipped" >&2
fi
# yq validation on staged .yaml/.yml files
if command -v yq &>/dev/null; then
while IFS= read -r f; do
[[ -f "$f" ]] && yq eval '.' "$f" > /dev/null
done < <(echo "$staged" | grep -E '\.(yaml|yml)$' || true)
else
echo "Warning: yq not installed — YAML validation skipped" >&2
fi
# SKILL.md frontmatter: must have name: and description:
while IFS= read -r f; do
if [[ -f "$f" ]]; then
if ! grep -q '^name:' "$f" || ! grep -q '^description' "$f"; then
echo "ERROR: $f is missing required frontmatter fields (name: and description:)" >&2
exit 1
fi
fi
done < <(echo "$staged" | grep 'SKILL\.md$' || true)
BLOCK
}
write_block "$PRE_COMMIT_HOOK" "$(pre_commit_block)"
echo "Updated: $PRE_COMMIT_HOOK (shellcheck + jq/yq + SKILL.md validation)"
# --- pre-push: test suite + manifest cross-reference ---
PRE_PUSH_HOOK="$TARGET/.git/hooks/pre-push"
ensure_hook "$PRE_PUSH_HOOK"
pre_push_block() {
local script_dir="$SCRIPT_DIR"
cat <<BLOCK
HOOKS_SCRIPT_DIR="$script_dir"
REPO_ROOT="\$(git rev-parse --show-toplevel)"
bash "\$REPO_ROOT/tests/run-tests.sh"
echo "Checking manifests..."
bash "\$HOOKS_SCRIPT_DIR/check-manifests.sh" "\$REPO_ROOT"
BLOCK
}
write_block "$PRE_PUSH_HOOK" "$(pre_push_block)"
echo "Updated: $PRE_PUSH_HOOK (test suite + manifest check)"
echo ""
echo "Done. Hooks installed in $TARGET/.git/hooks/"
echo "To skip on a single push: git push --no-verify"
-36
View File
@@ -1,36 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
# Installs @neuledge/context globally at a pinned version.
# Usage: setup-neuledge-context.sh [VERSION]
# VERSION — npm version string (default: 1.2.0)
# Idempotent: skips install if already at the target version.
TARGET_VERSION="${1:-1.2.0}"
current_version() {
context --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || echo ""
}
CURRENT="$(current_version)"
if [ "$CURRENT" = "$TARGET_VERSION" ]; then
echo "Already at @neuledge/context@${TARGET_VERSION} — nothing to do."
exit 0
fi
if [ -n "$CURRENT" ]; then
echo "Upgrading @neuledge/context from ${CURRENT} to ${TARGET_VERSION}..."
else
echo "Installing @neuledge/context@${TARGET_VERSION}..."
fi
npm install -g "@neuledge/context@${TARGET_VERSION}"
INSTALLED="$(current_version)"
if [ "$INSTALLED" = "$TARGET_VERSION" ]; then
echo "Done: @neuledge/context@${TARGET_VERSION} installed."
else
echo "Error: expected version ${TARGET_VERSION} but got '${INSTALLED}'" >&2
exit 1
fi
-151
View File
@@ -1,151 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
SCRIPT="$REPO_ROOT/scripts/setup-gitleaks.sh"
PASS=0
FAIL=0
pass() { echo " PASS: $1"; PASS=$((PASS + 1)); }
fail() { echo " FAIL: $1"; FAIL=$((FAIL + 1)); }
# Fake gitleaks binary — prevents the install step from running during tests
FAKE_BIN="$(mktemp -d)"
trap 'rm -rf "$FAKE_BIN"' EXIT
printf '#!/bin/sh\necho "gitleaks fake"\n' > "$FAKE_BIN/gitleaks"
chmod +x "$FAKE_BIN/gitleaks"
export PATH="$FAKE_BIN:$PATH"
# Helper: create an isolated git repo in a temp dir
make_repo() {
local dir
dir="$(mktemp -d)"
git -C "$dir" init -q
echo "$dir"
}
# Helper: run setup script against a target repo; capture output; always return exit code
run_setup() {
local target="$1"
bash "$SCRIPT" "$target" 2>&1
}
# --- 1. Rejects non-git directory ---
echo ""
echo "--- rejects non-git directory ---"
NON_GIT="$(mktemp -d)"
trap 'rm -rf "$NON_GIT"' EXIT
if bash "$SCRIPT" "$NON_GIT" >/dev/null 2>&1; then
fail "exited 0 for non-git directory — expected exit 1"
else
pass "exits non-zero for non-git directory"
fi
# --- 2. Config deployed ---
echo ""
echo "--- config deployed to target repo ---"
REPO="$(make_repo)"
trap 'rm -rf "$REPO"' EXIT
run_setup "$REPO" > /dev/null
if [ -f "$REPO/.gitleaks.toml" ]; then
pass ".gitleaks.toml created in target repo"
else
fail ".gitleaks.toml missing from target repo"
fi
if diff -q "$REPO_ROOT/scripts/gitleaks.toml" "$REPO/.gitleaks.toml" > /dev/null 2>&1; then
pass ".gitleaks.toml matches the template"
else
fail ".gitleaks.toml content differs from template"
fi
# --- 3. Hook created from scratch ---
echo ""
echo "--- hook created when none exists ---"
REPO2="$(make_repo)"
trap 'rm -rf "$REPO2"' EXIT
run_setup "$REPO2" > /dev/null
HOOK="$REPO2/.git/hooks/pre-commit"
if [ -f "$HOOK" ]; then
pass "pre-commit hook created"
else
fail "pre-commit hook not created"
fi
if [ -x "$HOOK" ]; then
pass "pre-commit hook is executable"
else
fail "pre-commit hook is not executable"
fi
if head -1 "$HOOK" | grep -q "^#!"; then
pass "pre-commit hook has a shebang"
else
fail "pre-commit hook missing shebang"
fi
if grep -q "gitleaks git --staged" "$HOOK"; then
pass "pre-commit hook contains gitleaks command"
else
fail "pre-commit hook missing gitleaks command"
fi
if grep -q "# managed by setup-gitleaks.sh" "$HOOK"; then
pass "pre-commit hook contains idempotency marker"
else
fail "pre-commit hook missing idempotency marker"
fi
# --- 4. Appends to existing hook; existing content retained ---
echo ""
echo "--- appends to existing hook; prior content retained ---"
REPO3="$(make_repo)"
trap 'rm -rf "$REPO3"' EXIT
HOOK3="$REPO3/.git/hooks/pre-commit"
printf '#!/usr/bin/env bash\nnpm test\n' > "$HOOK3"
chmod +x "$HOOK3"
run_setup "$REPO3" > /dev/null
if grep -q "npm test" "$HOOK3"; then
pass "existing hook content retained after append"
else
fail "existing hook content lost after append"
fi
if grep -q "gitleaks git --staged" "$HOOK3"; then
pass "gitleaks block appended to existing hook"
else
fail "gitleaks block missing after append"
fi
# --- 5. Second run replaces stale block; existing content still retained ---
echo ""
echo "--- second run replaces stale block; existing content still retained ---"
# Corrupt the gitleaks block to simulate stale content from an older version
sed -i 's/gitleaks git --staged/gitleaks protect --staged/' "$HOOK3"
run_setup "$REPO3" > /dev/null
if grep -q "npm test" "$HOOK3"; then
pass "existing content retained after block replacement"
else
fail "existing content lost after block replacement"
fi
marker_count="$(grep -c "# managed by setup-gitleaks.sh" "$HOOK3")"
if [ "$marker_count" -eq 1 ]; then
pass "gitleaks block appears exactly once after second run"
else
fail "gitleaks block duplicated — found $marker_count occurrences of marker"
fi
if grep -q "gitleaks git --staged" "$HOOK3"; then
pass "stale gitleaks command replaced with current command"
else
fail "stale gitleaks command not replaced — block was skipped, not updated"
fi
# --- 6. Third run still idempotent ---
echo ""
echo "--- repeated runs stay idempotent ---"
run_setup "$REPO3" > /dev/null
run_setup "$REPO3" > /dev/null
marker_count="$(grep -c "# managed by setup-gitleaks.sh" "$HOOK3")"
if [ "$marker_count" -eq 1 ]; then
pass "gitleaks block still appears exactly once after four total runs"
else
fail "gitleaks block duplicated — found $marker_count occurrences after four runs"
fi
echo ""
echo "Results: $PASS passed, $FAIL failed"
[[ $FAIL -eq 0 ]]
-210
View File
@@ -1,210 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
SCRIPT="$REPO_ROOT/scripts/setup-hooks.sh"
PASS=0
FAIL=0
pass() { echo " PASS: $1"; PASS=$((PASS + 1)); }
fail() { echo " FAIL: $1"; FAIL=$((FAIL + 1)); }
# Helper: make a bare git repo with no hooks yet
make_repo() {
local dir
dir="$(mktemp -d)"
git -C "$dir" init -q
echo "$dir"
}
# Fake binaries for tools we don't want to install-check during tests
FAKE_BIN="$(mktemp -d)"
trap 'rm -rf "$FAKE_BIN"' EXIT
for tool in shellcheck jq yq; do
printf '#!/bin/sh\necho "fake %s"\n' "$tool" > "$FAKE_BIN/$tool"
chmod +x "$FAKE_BIN/$tool"
done
export PATH="$FAKE_BIN:$PATH"
# --- 1. Rejects non-git directory ---
echo ""
echo "--- rejects non-git directory ---"
NON_GIT="$(mktemp -d)"
trap 'rm -rf "$NON_GIT"' EXIT
if bash "$SCRIPT" "$NON_GIT" > /dev/null 2>&1; then
fail "exited 0 for non-git directory — expected exit 1"
else
pass "exits non-zero for non-git directory"
fi
# --- 2. Creates commit-msg hook ---
echo ""
echo "--- creates commit-msg hook ---"
REPO="$(make_repo)"
trap 'rm -rf "$REPO"' EXIT
bash "$SCRIPT" "$REPO" > /dev/null 2>&1
HOOK="$REPO/.git/hooks/commit-msg"
if [[ -f "$HOOK" ]]; then
pass "commit-msg hook file created"
else
fail "commit-msg hook not created"
fi
if [[ -x "$HOOK" ]]; then
pass "commit-msg hook is executable"
else
fail "commit-msg hook is not executable"
fi
if grep -q "# managed by setup-hooks.sh" "$HOOK"; then
pass "commit-msg hook contains idempotency marker"
else
fail "commit-msg hook missing idempotency marker"
fi
# --- 3. commit-msg hook validates conventional commits ---
echo ""
echo "--- commit-msg hook: valid message passes ---"
REPO2="$(make_repo)"
trap 'rm -rf "$REPO2"' EXIT
bash "$SCRIPT" "$REPO2" > /dev/null 2>&1
HOOK2="$REPO2/.git/hooks/commit-msg"
TMPFILE="$(mktemp)"
trap 'rm -f "$TMPFILE"' EXIT
for valid_msg in "feat: add validation" "fix(core): correct path resolution" "chore!: drop python dep" "docs: update readme" "refactor(hooks): extract marker logic"; do
echo "$valid_msg" > "$TMPFILE"
if bash "$HOOK2" "$TMPFILE" > /dev/null 2>&1; then
pass "commit-msg hook accepts: $valid_msg"
else
fail "commit-msg hook wrongly rejected: $valid_msg"
fi
done
echo ""
echo "--- commit-msg hook: invalid message is rejected ---"
for invalid_msg in "added some stuff" "WIP" "Fix the thing" "FEAT: bad case" "feat bad colon"; do
echo "$invalid_msg" > "$TMPFILE"
if bash "$HOOK2" "$TMPFILE" > /dev/null 2>&1; then
fail "commit-msg hook wrongly accepted: $invalid_msg"
else
pass "commit-msg hook rejects: $invalid_msg"
fi
done
# --- 4. Appends pre-commit validation block ---
echo ""
echo "--- appends validation block to pre-commit hook ---"
REPO3="$(make_repo)"
trap 'rm -rf "$REPO3"' EXIT
bash "$SCRIPT" "$REPO3" > /dev/null 2>&1
PRE_COMMIT="$REPO3/.git/hooks/pre-commit"
if [[ -f "$PRE_COMMIT" ]]; then
pass "pre-commit hook created"
else
fail "pre-commit hook not created"
fi
if grep -q "shellcheck" "$PRE_COMMIT"; then
pass "pre-commit hook contains shellcheck"
else
fail "pre-commit hook missing shellcheck"
fi
if grep -q "jq" "$PRE_COMMIT"; then
pass "pre-commit hook contains jq"
else
fail "pre-commit hook missing jq"
fi
if grep -q "SKILL.md" "$PRE_COMMIT"; then
pass "pre-commit hook contains SKILL.md frontmatter check"
else
fail "pre-commit hook missing SKILL.md frontmatter check"
fi
# --- 5. Creates pre-push hook ---
echo ""
echo "--- creates pre-push hook ---"
REPO4="$(make_repo)"
trap 'rm -rf "$REPO4"' EXIT
bash "$SCRIPT" "$REPO4" > /dev/null 2>&1
PUSH_HOOK="$REPO4/.git/hooks/pre-push"
if [[ -f "$PUSH_HOOK" ]]; then
pass "pre-push hook created"
else
fail "pre-push hook not created"
fi
if [[ -x "$PUSH_HOOK" ]]; then
pass "pre-push hook is executable"
else
fail "pre-push hook is not executable"
fi
if grep -q "check-manifests" "$PUSH_HOOK"; then
pass "pre-push hook calls check-manifests.sh"
else
fail "pre-push hook missing check-manifests.sh call"
fi
if grep -q "run-tests.sh" "$PUSH_HOOK"; then
pass "pre-push hook calls run-tests.sh"
else
fail "pre-push hook missing run-tests.sh call"
fi
# --- 6. Idempotent: second run replaces each block exactly once ---
echo ""
echo "--- idempotent: second run does not duplicate blocks ---"
REPO5="$(make_repo)"
trap 'rm -rf "$REPO5"' EXIT
bash "$SCRIPT" "$REPO5" > /dev/null 2>&1
bash "$SCRIPT" "$REPO5" > /dev/null 2>&1
bash "$SCRIPT" "$REPO5" > /dev/null 2>&1
for hook_file in "$REPO5/.git/hooks/commit-msg" "$REPO5/.git/hooks/pre-commit" "$REPO5/.git/hooks/pre-push"; do
count=$(grep -c "# managed by setup-hooks.sh" "$hook_file" || true)
hook_name="$(basename "$hook_file")"
if [[ "$count" -eq 1 ]]; then
pass "idempotent: $hook_name marker appears exactly once after 3 runs"
else
fail "idempotent: $hook_name marker appears $count times — block duplicated"
fi
if [[ -x "$hook_file" ]]; then
pass "idempotent: $hook_name remains executable after 3 runs"
else
fail "idempotent: $hook_name lost executable bit after repeated runs"
fi
done
# --- 7. Setup installs tools; no "not installed" warnings when tools present ---
echo ""
echo "--- setup does not warn when tools are available ---"
REPO6="$(make_repo)"
trap 'rm -rf "$REPO6"' EXIT
output6="$(bash "$SCRIPT" "$REPO6" 2>&1)"
for tool in shellcheck jq yq; do
if echo "$output6" | grep -qi "$tool not installed\|$tool.*not found"; then
fail "setup warned about missing $tool — should install or already be present"
else
pass "setup emits no 'not installed' warning for: $tool (present or installed)"
fi
done
echo ""
echo "--- pre-commit hook retains runtime fallback for missing tools ---"
PRE_COMMIT6="$REPO6/.git/hooks/pre-commit"
for tool in shellcheck jq yq; do
if grep -q "Warning:.*$tool\|$tool.*not installed\|$tool.*skipped" "$PRE_COMMIT6"; then
pass "pre-commit hook has runtime fallback for missing: $tool"
else
fail "pre-commit hook missing runtime fallback for: $tool"
fi
done
echo ""
echo "--- setup exits 0 when tools are present ---"
REPO7="$(make_repo)"
trap 'rm -rf "$REPO7"' EXIT
if bash "$SCRIPT" "$REPO7" > /dev/null 2>&1; then
pass "setup exits 0 when tools are present"
else
fail "setup exited non-zero unexpectedly"
fi
echo ""
echo "Results: $PASS passed, $FAIL failed"
[[ $FAIL -eq 0 ]]