feat(skills): install deepeval skill from confident-ai/deepeval
Adds the deepeval eval-loop skill via `npx skills add` with skills-lock.json for reproducible reinstalls. Symlinked to Claude Code via .claude/skills/. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016z2ZFYHQCex8yZAMVMTZzZ
This commit is contained in:
38
.agents/skills/deepeval/templates/metrics.py
Normal file
38
.agents/skills/deepeval/templates/metrics.py
Normal file
@@ -0,0 +1,38 @@
|
||||
from deepeval.metrics import (
|
||||
AnswerRelevancyMetric,
|
||||
ContextualRelevancyMetric,
|
||||
StepEfficiencyMetric,
|
||||
TaskCompletionMetric,
|
||||
)
|
||||
|
||||
|
||||
# Keep metrics in one module so eval files stay focused on app execution.
|
||||
# Reuse existing project metrics and thresholds before adding new ones.
|
||||
SINGLE_TURN_TRACE_METRICS = [
|
||||
TaskCompletionMetric(),
|
||||
StepEfficiencyMetric(),
|
||||
]
|
||||
|
||||
SINGLE_TURN_NO_TRACING_METRICS = [
|
||||
AnswerRelevancyMetric(),
|
||||
]
|
||||
|
||||
MULTI_TURN_METRICS = []
|
||||
|
||||
# Component-level metrics are span-specific. Do not create one shared
|
||||
# COMPONENT_METRICS list for the whole app. Name each list after the exact
|
||||
# component/span it evaluates, then attach it with either:
|
||||
# - next_agent_span / next_llm_span / next_tool_span / next_retriever_span
|
||||
# - @observe(metrics=[...]) when the integration or manual instrumentation
|
||||
# creates the component span directly.
|
||||
RETRIEVER_SPAN_METRICS = [
|
||||
ContextualRelevancyMetric(),
|
||||
]
|
||||
|
||||
GENERATOR_LLM_SPAN_METRICS = [
|
||||
AnswerRelevancyMetric(),
|
||||
]
|
||||
|
||||
TOOL_SPAN_METRICS = []
|
||||
|
||||
PLANNER_AGENT_SPAN_METRICS = []
|
||||
28
.agents/skills/deepeval/templates/test_multi_turn_e2e.py
Normal file
28
.agents/skills/deepeval/templates/test_multi_turn_e2e.py
Normal file
@@ -0,0 +1,28 @@
|
||||
from importlib import import_module
|
||||
|
||||
import pytest
|
||||
|
||||
from deepeval import assert_test
|
||||
from deepeval.dataset import EvaluationDataset
|
||||
from deepeval.simulator import ConversationSimulator
|
||||
|
||||
from metrics import MULTI_TURN_METRICS
|
||||
|
||||
MAX_TURNS = 10
|
||||
ai_app = import_module("ai_app")
|
||||
|
||||
|
||||
simulator = ConversationSimulator(model_callback=ai_app.chatbot_callback)
|
||||
dataset = EvaluationDataset()
|
||||
dataset.add_goldens_from_json_file(file_path="tests/evals/.dataset.json")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"test_case",
|
||||
simulator.simulate(
|
||||
conversational_goldens=dataset.goldens,
|
||||
max_user_simulations=MAX_TURNS,
|
||||
),
|
||||
)
|
||||
def test_multi_turn(test_case):
|
||||
assert_test(test_case=test_case, metrics=MULTI_TURN_METRICS)
|
||||
@@ -0,0 +1,32 @@
|
||||
from importlib import import_module
|
||||
|
||||
import pytest
|
||||
|
||||
from deepeval import assert_test
|
||||
from deepeval.dataset import EvaluationDataset, Golden
|
||||
from deepeval.test_case import LLMTestCase
|
||||
|
||||
from metrics import SINGLE_TURN_NO_TRACING_METRICS
|
||||
|
||||
|
||||
ai_app = import_module("ai_app")
|
||||
|
||||
|
||||
dataset = EvaluationDataset()
|
||||
dataset.add_goldens_from_json_file(file_path="tests/evals/.dataset.json")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("golden", dataset.goldens)
|
||||
def test_single_turn_no_tracing(golden: Golden):
|
||||
actual_output = ai_app.run_ai_app(golden.input)
|
||||
test_case = LLMTestCase(
|
||||
input=golden.input,
|
||||
actual_output=actual_output,
|
||||
expected_output=getattr(golden, "expected_output", None),
|
||||
context=getattr(golden, "context", None),
|
||||
retrieval_context=getattr(golden, "retrieval_context", None),
|
||||
)
|
||||
assert_test(
|
||||
test_case=test_case,
|
||||
metrics=SINGLE_TURN_NO_TRACING_METRICS,
|
||||
)
|
||||
@@ -0,0 +1,21 @@
|
||||
from importlib import import_module
|
||||
|
||||
import pytest
|
||||
|
||||
from deepeval import assert_test
|
||||
from deepeval.dataset import EvaluationDataset, Golden
|
||||
|
||||
from metrics import SINGLE_TURN_TRACE_METRICS
|
||||
|
||||
|
||||
ai_app = import_module("ai_app")
|
||||
|
||||
|
||||
dataset = EvaluationDataset()
|
||||
dataset.add_goldens_from_json_file(file_path="tests/evals/.dataset.json")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("golden", dataset.goldens)
|
||||
def test_single_turn_tracing(golden: Golden):
|
||||
ai_app.run_traced_ai_app(golden.input)
|
||||
assert_test(golden=golden, metrics=SINGLE_TURN_TRACE_METRICS)
|
||||
Reference in New Issue
Block a user