Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
159 changes: 151 additions & 8 deletions evolution/core/external_importers.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,8 @@
Supported sources:
- Claude Code (~/.claude/history.jsonl) — user inputs only
- GitHub Copilot (~/.copilot/session-state/*/events.jsonl) — full conversations
- Hermes Agent (~/.hermes/sessions/*.json) — user + assistant + tool context
- Hermes Agent (state.db, or legacy ~/.hermes/sessions/*.json) — user +
assistant + tool context

Usage as standalone CLI:
python -m evolution.core.external_importers \\
Expand All @@ -23,8 +24,10 @@
"""

import json
import os
import re
import random
import sqlite3
from pathlib import Path
from typing import Optional

Expand Down Expand Up @@ -332,22 +335,150 @@ def _parse_copilot_events(


class HermesSessionImporter:
"""Import conversations from Hermes Agent session files.
"""Import conversations from the Hermes Agent session store.

Hermes stores session transcripts as JSON files in ~/.hermes/sessions/.
Each file contains an OpenAI-format message list with user, assistant,
and tool messages — providing richer signal than Claude Code (user-only)
or Copilot (user+assistant without tool context).
Modern Hermes (>= v0.15) stores every transcript in a single SQLite
database (``state.db``) with ``sessions`` and ``messages`` tables — not as
one JSON file per session. Older builds used ``~/.hermes/sessions/*.json``.
This importer prefers the SQLite store and falls back to the legacy JSON
layout so both generations keep working.

This mines user messages paired with the assistant's final response,
giving the LLM judge both the task and how it was actually handled.
"""

# Legacy layout (pre-SQLite Hermes builds).
SESSION_DIR = Path.home() / ".hermes" / "sessions"

# Explicit SQLite store override. ``None`` means auto-discover; set it to a
# concrete path to pin the database (and to isolate tests from whatever
# Hermes install happens to exist on the machine running them).
STATE_DB: Optional[Path] = None

# Cron-injected preamble — agent-generated scaffolding, not a user task.
_CRON_PREAMBLE = "[IMPORTANT: You are running as a scheduled cron job"

@staticmethod
def _candidate_db_paths() -> list[Path]:
"""Return plausible locations of the Hermes ``state.db``, best first.

An explicit :attr:`STATE_DB` short-circuits discovery. Otherwise honors
``HERMES_HOME``, then the per-platform defaults (Windows
``%LOCALAPPDATA%\\hermes``, XDG data dir, macOS Application Support,
``~/.hermes``).
"""
override = HermesSessionImporter.STATE_DB
if override is not None:
override = Path(override).expanduser()
return [override] if override.is_file() else []

candidates: list[Path] = []

hermes_home = os.environ.get("HERMES_HOME")
if hermes_home:
candidates.append(Path(hermes_home).expanduser())

home = Path.home()
local_appdata = os.environ.get("LOCALAPPDATA")
if local_appdata:
candidates.append(Path(local_appdata) / "hermes")
candidates.append(home / "AppData" / "Local" / "hermes") # Windows
candidates.append(home / ".local" / "share" / "hermes") # XDG
candidates.append(home / "Library" / "Application Support" / "hermes") # macOS
candidates.append(home / ".hermes") # legacy / custom

seen: set[Path] = set()
paths: list[Path] = []
for base in candidates:
db = base / "state.db"
if db in seen:
continue
seen.add(db)
if db.is_file():
paths.append(db)
return paths

@staticmethod
def _extract_from_sqlite(db_path: Path, limit: int = 0) -> list[dict]:
"""Mine user/assistant pairs out of a Hermes ``state.db``."""
messages: list[dict] = []

# Read-only URI so a live Hermes process is never disturbed.
try:
conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True)
except sqlite3.Error:
return []

try:
conn.row_factory = sqlite3.Row
cols = {r[1] for r in conn.execute("PRAGMA table_info(messages)")}
if not {"session_id", "role", "content"} <= cols:
return []

order_col = "id" if "id" in cols else "rowid"
# Only real transcript turns; skip compacted/summary rows when the
# column exists so we don't train on compression artifacts.
where = "role IN ('user','assistant')"
if "compacted" in cols:
where += " AND COALESCE(compacted, 0) = 0"

rows = conn.execute(
f"SELECT session_id, role, content FROM messages "
f"WHERE {where} ORDER BY session_id, {order_col}"
).fetchall()
except sqlite3.Error:
return []
finally:
conn.close()

# Group by session, preserving order, then pair user -> next assistant.
by_session: dict[str, list[sqlite3.Row]] = {}
for row in rows:
by_session.setdefault(row["session_id"], []).append(row)

for session_id, msg_list in by_session.items():
for i, msg in enumerate(msg_list):
if msg["role"] != "user":
continue
user_text = msg["content"] or ""
if not isinstance(user_text, str) or len(user_text) < 10:
continue
if user_text.lstrip().startswith(HermesSessionImporter._CRON_PREAMBLE):
continue
if _contains_secret(user_text):
continue

assistant_text = ""
for j in range(i + 1, len(msg_list)):
if msg_list[j]["role"] == "assistant":
content = msg_list[j]["content"] or ""
if content:
assistant_text = content
break
elif msg_list[j]["role"] == "user":
break

if assistant_text and _contains_secret(assistant_text):
continue

messages.append({
"source": "hermes",
"task_input": user_text,
"assistant_response": assistant_text,
"session_id": session_id,
})

if limit and len(messages) >= limit:
return messages

return messages

@staticmethod
def extract_messages(limit: int = 0) -> list[dict]:
"""Read user/assistant pairs from Hermes session files.
"""Read user/assistant pairs from the Hermes session store.

Tries the SQLite ``state.db`` first (modern Hermes), then the legacy
``~/.hermes/sessions/*.json`` layout.

Args:
limit: Maximum messages to return (0 = no limit).
Expand All @@ -356,6 +487,18 @@ def extract_messages(limit: int = 0) -> list[dict]:
List of dicts with keys: source, task_input, assistant_response,
session_id.
"""
for db_path in HermesSessionImporter._candidate_db_paths():
found = HermesSessionImporter._extract_from_sqlite(db_path, limit=limit)
if found:
return found

# An explicit existing STATE_DB pins the source (especially in tests). If
# that DB yields no usable messages, do not fall through into the user's
# real legacy session directory and contaminate the result. A missing
# override still allows legacy fallback for compatibility.
if HermesSessionImporter.STATE_DB is not None and Path(HermesSessionImporter.STATE_DB).expanduser().is_file():
return []

if not HermesSessionImporter.SESSION_DIR.exists():
return []

Expand All @@ -369,7 +512,7 @@ def extract_messages(limit: int = 0) -> list[dict]:
for session_file in session_files:
try:
data = json.loads(session_file.read_text())
except (json.JSONDecodeError, OSError):
except (UnicodeDecodeError, json.JSONDecodeError, OSError):
continue

msg_list = data.get("messages", [])
Expand Down
35 changes: 30 additions & 5 deletions evolution/skills/evolve_skill.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,21 @@
console = Console()


def _gepa_metric(*args, **kwargs) -> float:
"""Arity-tolerant wrapper around :func:`skill_fitness_metric`.

DSPy optimizers call metrics with varying signatures — GEPA passes
``(gold, pred, trace, pred_name, pred_trace)`` while Evaluate passes
``(gold, pred)``. ``skill_fitness_metric`` only accepts three positional
args, so calling it directly raises ``TypeError`` under GEPA. Normalize
here instead of changing the metric's public signature.
"""
example = args[0] if len(args) > 0 else kwargs.get("example") or kwargs.get("gold")
prediction = args[1] if len(args) > 1 else kwargs.get("prediction") or kwargs.get("pred")
trace = args[2] if len(args) > 2 else kwargs.get("trace")
return skill_fitness_metric(example, prediction, trace)


def evolve(
skill_name: str,
iterations: int = 10,
Expand Down Expand Up @@ -118,8 +133,13 @@ def evolve(
# ── 3. Validate constraints on baseline ─────────────────────────────
console.print(f"\n[bold]Validating baseline constraints[/bold]")
validator = ConstraintValidator(config)
baseline_constraints = validator.validate_all(skill["body"], "skill")
# Validate the FULL skill file, not just the body: _check_skill_structure
# asserts YAML frontmatter exists, and load_skill() has already stripped it
# from skill["body"] — passing the body guarantees a false failure, which
# would later block a perfectly good evolved skill from deploying.
baseline_constraints = validator.validate_all(skill["raw"], "skill")
all_pass = True

for c in baseline_constraints:
icon = "✓" if c.passed else "✗"
color = "green" if c.passed else "red"
Expand Down Expand Up @@ -153,9 +173,14 @@ def evolve(
start_time = time.time()

try:
# dspy>=3.x GEPA has no `max_steps`; `max_full_evals` is the closest
# analogue to "iterations" (budgeted full passes over the valset), and
# it requires an explicit reflection LM to propose mutations.
reflection_lm = dspy.LM(optimizer_model, temperature=1.0, max_tokens=8000)
optimizer = dspy.GEPA(
metric=skill_fitness_metric,
max_steps=iterations,
metric=_gepa_metric,
max_full_evals=iterations,
reflection_lm=reflection_lm,
)

optimized_module = optimizer.compile(
Expand All @@ -167,7 +192,7 @@ def evolve(
# Fall back to MIPROv2 if GEPA isn't available in this DSPy version
console.print(f"[yellow]GEPA not available ({e}), falling back to MIPROv2[/yellow]")
optimizer = dspy.MIPROv2(
metric=skill_fitness_metric,
metric=_gepa_metric,
auto="light",
)
optimized_module = optimizer.compile(
Expand All @@ -185,7 +210,7 @@ def evolve(

# ── 7. Validate evolved skill ───────────────────────────────────────
console.print(f"\n[bold]Validating evolved skill[/bold]")
evolved_constraints = validator.validate_all(evolved_body, "skill", baseline_text=skill["body"])
evolved_constraints = validator.validate_all(evolved_full, "skill", baseline_text=skill["raw"])
all_pass = True
for c in evolved_constraints:
icon = "✓" if c.passed else "✗"
Expand Down
42 changes: 31 additions & 11 deletions evolution/skills/skill_module.py
Original file line number Diff line number Diff line change
Expand Up @@ -84,33 +84,53 @@ def find_skill(skill_name: str, hermes_agent_path: Path) -> Optional[Path]:
class SkillModule(dspy.Module):
"""A DSPy module that wraps a skill file for optimization.

The skill text (body) is the parameter that GEPA optimizes.
The skill text is the parameter that GEPA optimizes. Critically, it must
live in the predictor's **signature instructions**, not in an InputField:
DSPy optimizers mutate signature instructions and leave input *values*
alone. Wiring the skill as an InputField (the original design) meant GEPA
happily proposed improved variants that were then silently discarded,
producing an "evolved" skill byte-identical to the baseline.

On each forward pass, the module:
1. Uses the skill text as instructions
1. Uses the skill text as the predictor's instructions
2. Processes the task input
3. Returns the agent's response
"""

class TaskWithSkill(dspy.Signature):
"""Complete a task following the provided skill instructions.

You are an AI agent following specific skill instructions to complete a task.
Read the skill instructions carefully and follow the procedure described.
You are an AI agent following specific skill instructions to complete a
task. Read the skill instructions carefully and follow the procedure
described.
"""
skill_instructions: str = dspy.InputField(desc="The skill instructions to follow")
task_input: str = dspy.InputField(desc="The task to complete")
output: str = dspy.OutputField(desc="Your response following the skill instructions")

def __init__(self, skill_text: str):
super().__init__()
self.skill_text = skill_text
self.predictor = dspy.ChainOfThought(self.TaskWithSkill)
# Seed the signature's instructions with the skill body so it becomes
# the optimizable parameter GEPA mutates.
self.predictor = dspy.ChainOfThought(
self.TaskWithSkill.with_instructions(skill_text)
)

@property
def skill_text(self) -> str:
"""Current skill text, read back out of the signature instructions.

After ``optimizer.compile(...)`` this returns the *evolved* text, which
is what callers persist to disk.
"""
# dspy.ChainOfThought wraps an inner Predict as `.predict`; GEPA names
# the component "predictor.predict". Fall back to the outer signature
# for other module shapes / DSPy versions.
inner = getattr(self.predictor, "predict", None)
sig = getattr(inner, "signature", None) or getattr(self.predictor, "signature", None)
return getattr(sig, "instructions", "") or ""

def forward(self, task_input: str) -> dspy.Prediction:
result = self.predictor(
skill_instructions=self.skill_text,
task_input=task_input,
)
result = self.predictor(task_input=task_input)
return dspy.Prediction(output=result.output)


Expand Down
5 changes: 5 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,11 @@ dev = [
darwinian = [
"darwinian-evolver",
]
# The MIPROv2 fallback in evolve_skill.py needs optuna; without it a GEPA
# failure turns into ImportError instead of a working fallback.
mipro = [
"optuna>=3.0",
]

[project.urls]
Homepage = "https://github.com/NousResearch/hermes-agent-self-evolution"
Expand Down
12 changes: 10 additions & 2 deletions tests/core/test_config_repo_path.py
Original file line number Diff line number Diff line change
Expand Up @@ -46,8 +46,16 @@ def test_resolve_honors_explicit_path_without_default(tmp_path, monkeypatch):
def test_resolve_expands_user_home(monkeypatch):
from evolution.core.config import resolve_hermes_agent_path

monkeypatch.setenv("HOME", "/home/example")
assert resolve_hermes_agent_path("~/code/hermes-agent") == Path("/home/example/code/hermes-agent")
# Path.expanduser() reads HOME on POSIX but USERPROFILE on Windows, so set
# both — otherwise this test only exercises tilde expansion on POSIX and
# fails on Windows against the real user profile.
fake_home = Path("/home/example")
monkeypatch.setenv("HOME", str(fake_home))
monkeypatch.setenv("USERPROFILE", str(fake_home))
monkeypatch.delenv("HOMEDRIVE", raising=False)
monkeypatch.delenv("HOMEPATH", raising=False)

assert resolve_hermes_agent_path("~/code/hermes-agent") == fake_home / "code" / "hermes-agent"


def test_resolve_falls_back_to_env_var_when_no_override(tmp_path, monkeypatch):
Expand Down
Loading