feat: implement issue 0018 phase 1 — factory/write-skill bootstrap skill
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
68
.agents/evals/factory/write-skill/eval.yaml
Normal file
68
.agents/evals/factory/write-skill/eval.yaml
Normal file
@@ -0,0 +1,68 @@
|
||||
skill_name: write-skill
|
||||
|
||||
trigger_tests:
|
||||
- id: explicit-trigger-new-skill
|
||||
name: Explicit — new skill phrase
|
||||
query: "Write a new skill for handling database migrations"
|
||||
should_trigger: true
|
||||
|
||||
- id: implicit-trigger-no-phrase
|
||||
name: Implicit — no trigger phrase
|
||||
query: "I want to add a skill that automates our deploy process"
|
||||
should_trigger: true
|
||||
|
||||
- id: implicit-trigger-conversion
|
||||
name: Implicit — placeholder conversion
|
||||
query: "The grill-me skill is a Pocock placeholder, can you convert it to our standard?"
|
||||
should_trigger: true
|
||||
|
||||
- id: negative-trigger-upgrade
|
||||
name: Negative — existing skill fix
|
||||
query: "The tdd skill is producing wrong output, fix it"
|
||||
should_trigger: false
|
||||
|
||||
- id: negative-trigger-code-refactor
|
||||
name: Negative — code refactor
|
||||
query: "Refactor this module to use the new API client"
|
||||
should_trigger: false
|
||||
|
||||
- id: negative-trigger-write-eval
|
||||
name: Negative — eval request
|
||||
query: "Write evals for the diagnose skill"
|
||||
should_trigger: false
|
||||
|
||||
output_tests:
|
||||
- id: output-has-all-sections
|
||||
name: All 8 body sections present
|
||||
type: deterministic
|
||||
prompt: "Write a new skill for linting markdown files, category: implement"
|
||||
expected_output: A complete SKILL.md containing all 8 required body sections in order.
|
||||
assertions:
|
||||
- "Output contains '## Role'"
|
||||
- "Output contains '## When to use / When not to use'"
|
||||
- "Output contains '## Required inputs'"
|
||||
- "Output contains '## Constraints'"
|
||||
- "Output contains '## Process'"
|
||||
- "Output contains '## Output format'"
|
||||
- "Output contains '## Failure handling'"
|
||||
- "Output contains '## Self-check'"
|
||||
|
||||
- id: output-path-correct
|
||||
name: Output path and frontmatter fields correct
|
||||
type: deterministic
|
||||
prompt: "Write a new skill for linting markdown files, category: implement"
|
||||
expected_output: A SKILL.md with correct output path stated and all required frontmatter fields present.
|
||||
assertions:
|
||||
- "Output contains '.agents/skills/' in the stated output path"
|
||||
- "Output contains 'metadata:' and 'category:' in frontmatter"
|
||||
- "Output contains 'version:'"
|
||||
- "Output contains 'when:'"
|
||||
|
||||
- id: output-trigger-tested-before-body
|
||||
name: Trigger description tested before body content written
|
||||
type: llm-rubric
|
||||
prompt: "Write a new skill for summarising pull request diffs"
|
||||
expected_output: The skill presents a trigger description and tests it against at least 3 cases (explicit, implicit, negative) before proposing or writing any body section content.
|
||||
assertions:
|
||||
- "The skill proposes a trigger description and explicitly tests it against an explicit query, an implicit query, and a negative query before writing any body section"
|
||||
- "The skill walks through each body section individually and seeks confirmation before writing the file"
|
||||
Reference in New Issue
Block a user