diff --git a/evolution/core/fitness.py b/evolution/core/fitness.py index 04f2c78b..4b767887 100644 --- a/evolution/core/fitness.py +++ b/evolution/core/fitness.py @@ -115,6 +115,13 @@ def skill_fitness_metric(example: dspy.Example, prediction: dspy.Prediction, tra expected = getattr(example, "expected_behavior", "") or "" task = getattr(example, "task_input", "") or "" + # Dataset generators may emit rubrics as a list of bullet strings — + # coerce to text before keyword-overlap scoring. + if isinstance(expected, list): + expected = "\n".join(str(item) for item in expected) + if isinstance(agent_output, list): + agent_output = "\n".join(str(item) for item in agent_output) + if not agent_output.strip(): return 0.0 diff --git a/evolution/skills/evolve_skill.py b/evolution/skills/evolve_skill.py index 2a79a670..bad28565 100644 --- a/evolution/skills/evolve_skill.py +++ b/evolution/skills/evolve_skill.py @@ -118,7 +118,7 @@ def evolve( # ── 3. Validate constraints on baseline ───────────────────────────── console.print(f"\n[bold]Validating baseline constraints[/bold]") validator = ConstraintValidator(config) - baseline_constraints = validator.validate_all(skill["body"], "skill") + baseline_constraints = validator.validate_all(skill["raw"], "skill") all_pass = True for c in baseline_constraints: icon = "✓" if c.passed else "✗" @@ -152,10 +152,17 @@ def evolve( start_time = time.time() + # GEPA (dspy>=3.2) requires a 5-arg metric (gold, pred, trace, pred_name, + # pred_trace). It must return a plain float: dspy.Evaluate aggregates + # metric results with sum(), so dict/Prediction returns break full evals. + def _gepa_metric(gold, pred, trace=None, pred_name=None, pred_trace=None): + return skill_fitness_metric(gold, pred) + try: optimizer = dspy.GEPA( - metric=skill_fitness_metric, - max_steps=iterations, + metric=_gepa_metric, + max_full_evals=iterations, + reflection_lm=dspy.LM(optimizer_model, temperature=1.0, max_tokens=32000), ) optimized_module = optimizer.compile( @@ -179,13 +186,15 @@ def evolve( console.print(f"\n Optimization completed in {elapsed:.1f}s") # ── 6. Extract evolved skill text ─────────────────────────────────── - # The optimized module's instructions contain the evolved skill text - evolved_body = optimized_module.skill_text + # GEPA evolves the predictor's instruction (which is where SkillModule + # installs the skill body), so pull the mutated instruction back out of + # the optimized predictor's signature. + evolved_body = optimized_module.predictor.predict.signature.instructions evolved_full = reassemble_skill(skill["frontmatter"], evolved_body) # ── 7. Validate evolved skill ─────────────────────────────────────── console.print(f"\n[bold]Validating evolved skill[/bold]") - evolved_constraints = validator.validate_all(evolved_body, "skill", baseline_text=skill["body"]) + evolved_constraints = validator.validate_all(evolved_full, "skill", baseline_text=skill["raw"]) all_pass = True for c in evolved_constraints: icon = "✓" if c.passed else "✗" diff --git a/evolution/skills/skill_module.py b/evolution/skills/skill_module.py index 6d4d22ed..d78f7a81 100644 --- a/evolution/skills/skill_module.py +++ b/evolution/skills/skill_module.py @@ -84,33 +84,26 @@ def find_skill(skill_name: str, hermes_agent_path: Path) -> Optional[Path]: class SkillModule(dspy.Module): """A DSPy module that wraps a skill file for optimization. - The skill text (body) is the parameter that GEPA optimizes. - On each forward pass, the module: + The skill text (body) is the parameter that GEPA optimizes, so it is + installed as the predictor's *instruction* (GEPA mutates instructions + and demos — never plain input fields). On each forward pass, the module: 1. Uses the skill text as instructions 2. Processes the task input 3. Returns the agent's response """ class TaskWithSkill(dspy.Signature): - """Complete a task following the provided skill instructions. - - You are an AI agent following specific skill instructions to complete a task. - Read the skill instructions carefully and follow the procedure described. - """ - skill_instructions: str = dspy.InputField(desc="The skill instructions to follow") + """Placeholder instruction — replaced by the skill text in __init__.""" task_input: str = dspy.InputField(desc="The task to complete") output: str = dspy.OutputField(desc="Your response following the skill instructions") def __init__(self, skill_text: str): super().__init__() self.skill_text = skill_text - self.predictor = dspy.ChainOfThought(self.TaskWithSkill) + self.predictor = dspy.ChainOfThought(self.TaskWithSkill.with_instructions(skill_text)) def forward(self, task_input: str) -> dspy.Prediction: - result = self.predictor( - skill_instructions=self.skill_text, - task_input=task_input, - ) + result = self.predictor(task_input=task_input) return dspy.Prediction(output=result.output) diff --git a/pyproject.toml b/pyproject.toml index dae7f461..5735ce56 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -26,6 +26,7 @@ dependencies = [ dev = [ "pytest>=7.0", "pytest-asyncio>=0.21", + "optuna>=3.0", ] darwinian = [ "darwinian-evolver",