Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions evolution/core/fitness.py
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,13 @@ def skill_fitness_metric(example: dspy.Example, prediction: dspy.Prediction, tra
expected = getattr(example, "expected_behavior", "") or ""
task = getattr(example, "task_input", "") or ""

# Dataset generators may emit rubrics as a list of bullet strings —
# coerce to text before keyword-overlap scoring.
if isinstance(expected, list):
expected = "\n".join(str(item) for item in expected)
if isinstance(agent_output, list):
agent_output = "\n".join(str(item) for item in agent_output)

if not agent_output.strip():
return 0.0

Expand Down
21 changes: 15 additions & 6 deletions evolution/skills/evolve_skill.py
Original file line number Diff line number Diff line change
Expand Up @@ -118,7 +118,7 @@ def evolve(
# ── 3. Validate constraints on baseline ─────────────────────────────
console.print(f"\n[bold]Validating baseline constraints[/bold]")
validator = ConstraintValidator(config)
baseline_constraints = validator.validate_all(skill["body"], "skill")
baseline_constraints = validator.validate_all(skill["raw"], "skill")
all_pass = True
for c in baseline_constraints:
icon = "✓" if c.passed else "✗"
Expand Down Expand Up @@ -152,10 +152,17 @@ def evolve(

start_time = time.time()

# GEPA (dspy>=3.2) requires a 5-arg metric (gold, pred, trace, pred_name,
# pred_trace). It must return a plain float: dspy.Evaluate aggregates
# metric results with sum(), so dict/Prediction returns break full evals.
def _gepa_metric(gold, pred, trace=None, pred_name=None, pred_trace=None):
return skill_fitness_metric(gold, pred)

try:
optimizer = dspy.GEPA(
metric=skill_fitness_metric,
max_steps=iterations,
metric=_gepa_metric,
max_full_evals=iterations,
reflection_lm=dspy.LM(optimizer_model, temperature=1.0, max_tokens=32000),
)

optimized_module = optimizer.compile(
Expand All @@ -179,13 +186,15 @@ def evolve(
console.print(f"\n Optimization completed in {elapsed:.1f}s")

# ── 6. Extract evolved skill text ───────────────────────────────────
# The optimized module's instructions contain the evolved skill text
evolved_body = optimized_module.skill_text
# GEPA evolves the predictor's instruction (which is where SkillModule
# installs the skill body), so pull the mutated instruction back out of
# the optimized predictor's signature.
evolved_body = optimized_module.predictor.predict.signature.instructions
evolved_full = reassemble_skill(skill["frontmatter"], evolved_body)

# ── 7. Validate evolved skill ───────────────────────────────────────
console.print(f"\n[bold]Validating evolved skill[/bold]")
evolved_constraints = validator.validate_all(evolved_body, "skill", baseline_text=skill["body"])
evolved_constraints = validator.validate_all(evolved_full, "skill", baseline_text=skill["raw"])
all_pass = True
for c in evolved_constraints:
icon = "✓" if c.passed else "✗"
Expand Down
19 changes: 6 additions & 13 deletions evolution/skills/skill_module.py
Original file line number Diff line number Diff line change
Expand Up @@ -84,33 +84,26 @@ def find_skill(skill_name: str, hermes_agent_path: Path) -> Optional[Path]:
class SkillModule(dspy.Module):
"""A DSPy module that wraps a skill file for optimization.

The skill text (body) is the parameter that GEPA optimizes.
On each forward pass, the module:
The skill text (body) is the parameter that GEPA optimizes, so it is
installed as the predictor's *instruction* (GEPA mutates instructions
and demos — never plain input fields). On each forward pass, the module:
1. Uses the skill text as instructions
2. Processes the task input
3. Returns the agent's response
"""

class TaskWithSkill(dspy.Signature):
"""Complete a task following the provided skill instructions.

You are an AI agent following specific skill instructions to complete a task.
Read the skill instructions carefully and follow the procedure described.
"""
skill_instructions: str = dspy.InputField(desc="The skill instructions to follow")
"""Placeholder instruction — replaced by the skill text in __init__."""
task_input: str = dspy.InputField(desc="The task to complete")
output: str = dspy.OutputField(desc="Your response following the skill instructions")

def __init__(self, skill_text: str):
super().__init__()
self.skill_text = skill_text
self.predictor = dspy.ChainOfThought(self.TaskWithSkill)
self.predictor = dspy.ChainOfThought(self.TaskWithSkill.with_instructions(skill_text))

def forward(self, task_input: str) -> dspy.Prediction:
result = self.predictor(
skill_instructions=self.skill_text,
task_input=task_input,
)
result = self.predictor(task_input=task_input)
return dspy.Prediction(output=result.output)


Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@ dependencies = [
dev = [
"pytest>=7.0",
"pytest-asyncio>=0.21",
"optuna>=3.0",
]
darwinian = [
"darwinian-evolver",
Expand Down