Compare commits
1 Commits
q/327-1776
...
whip/322-1
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3563896f86 |
171
agent/memory/__init__.py
Normal file
171
agent/memory/__init__.py
Normal file
@@ -0,0 +1,171 @@
|
||||
"""Memory Backend Interface — pluggable cross-session user modeling.
|
||||
|
||||
Provides a common interface for memory backends that persist user
|
||||
preferences and patterns across sessions. Two implementations:
|
||||
|
||||
1. LocalBackend (default): SQLite-based, zero cloud dependency
|
||||
2. HonchoBackend (opt-in): Honcho AI-native memory, requires API key
|
||||
|
||||
Both are zero-overhead when disabled — the interface returns empty
|
||||
results and no writes occur.
|
||||
|
||||
Usage:
|
||||
from agent.memory import get_memory_backend
|
||||
|
||||
backend = get_memory_backend() # returns configured backend
|
||||
backend.store_preference("user", "prefers_python", "True")
|
||||
context = backend.query_context("user", "What does this user prefer?")
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import sqlite3
|
||||
import time
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from hermes_constants import get_hermes_home
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class MemoryEntry:
|
||||
"""A single memory entry."""
|
||||
key: str
|
||||
value: str
|
||||
user_id: str
|
||||
created_at: float = 0
|
||||
updated_at: float = 0
|
||||
metadata: Dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def __post_init__(self):
|
||||
now = time.time()
|
||||
if not self.created_at:
|
||||
self.created_at = now
|
||||
if not self.updated_at:
|
||||
self.updated_at = now
|
||||
|
||||
|
||||
class MemoryBackend(ABC):
|
||||
"""Abstract interface for memory backends."""
|
||||
|
||||
@abstractmethod
|
||||
def is_available(self) -> bool:
|
||||
"""Check if this backend is configured and usable."""
|
||||
|
||||
@abstractmethod
|
||||
def store(self, user_id: str, key: str, value: str, metadata: Dict = None) -> bool:
|
||||
"""Store a memory entry."""
|
||||
|
||||
@abstractmethod
|
||||
def retrieve(self, user_id: str, key: str) -> Optional[MemoryEntry]:
|
||||
"""Retrieve a single memory entry."""
|
||||
|
||||
@abstractmethod
|
||||
def query(self, user_id: str, query_text: str, limit: int = 10) -> List[MemoryEntry]:
|
||||
"""Query memories relevant to a text query."""
|
||||
|
||||
@abstractmethod
|
||||
def list_keys(self, user_id: str) -> List[str]:
|
||||
"""List all keys for a user."""
|
||||
|
||||
@abstractmethod
|
||||
def delete(self, user_id: str, key: str) -> bool:
|
||||
"""Delete a memory entry."""
|
||||
|
||||
@property
|
||||
@abstractmethod
|
||||
def backend_name(self) -> str:
|
||||
"""Human-readable backend name."""
|
||||
|
||||
@property
|
||||
@abstractmethod
|
||||
def is_cloud(self) -> bool:
|
||||
"""Whether this backend requires cloud connectivity."""
|
||||
|
||||
|
||||
class NullBackend(MemoryBackend):
|
||||
"""No-op backend when memory is disabled. Zero overhead."""
|
||||
|
||||
def is_available(self) -> bool:
|
||||
return True # always "available" as null
|
||||
|
||||
def store(self, user_id: str, key: str, value: str, metadata: Dict = None) -> bool:
|
||||
return True # no-op
|
||||
|
||||
def retrieve(self, user_id: str, key: str) -> Optional[MemoryEntry]:
|
||||
return None
|
||||
|
||||
def query(self, user_id: str, query_text: str, limit: int = 10) -> List[MemoryEntry]:
|
||||
return []
|
||||
|
||||
def list_keys(self, user_id: str) -> List[str]:
|
||||
return []
|
||||
|
||||
def delete(self, user_id: str, key: str) -> bool:
|
||||
return True
|
||||
|
||||
@property
|
||||
def backend_name(self) -> str:
|
||||
return "null (disabled)"
|
||||
|
||||
@property
|
||||
def is_cloud(self) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Singleton
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_backend: Optional[MemoryBackend] = None
|
||||
|
||||
|
||||
def get_memory_backend() -> MemoryBackend:
|
||||
"""Get the configured memory backend.
|
||||
|
||||
Priority:
|
||||
1. If HONCHO_API_KEY is set and honcho-ai is installed -> HonchoBackend
|
||||
2. If memory_backend config is 'local' -> LocalBackend
|
||||
3. Default -> NullBackend (zero overhead)
|
||||
"""
|
||||
global _backend
|
||||
if _backend is not None:
|
||||
return _backend
|
||||
|
||||
# Check config
|
||||
backend_type = os.getenv("HERMES_MEMORY_BACKEND", "").lower().strip()
|
||||
|
||||
if backend_type == "honcho" or os.getenv("HONCHO_API_KEY"):
|
||||
try:
|
||||
from agent.memory.honcho_backend import HonchoBackend
|
||||
backend = HonchoBackend()
|
||||
if backend.is_available():
|
||||
_backend = backend
|
||||
logger.info("Memory backend: Honcho (cloud)")
|
||||
return _backend
|
||||
except ImportError:
|
||||
logger.debug("Honcho not installed, falling back")
|
||||
|
||||
if backend_type == "local":
|
||||
try:
|
||||
from agent.memory.local_backend import LocalBackend
|
||||
_backend = LocalBackend()
|
||||
logger.info("Memory backend: Local (SQLite)")
|
||||
return _backend
|
||||
except Exception as e:
|
||||
logger.warning("Local backend failed: %s", e)
|
||||
|
||||
# Default: null (zero overhead)
|
||||
_backend = NullBackend()
|
||||
return _backend
|
||||
|
||||
|
||||
def reset_backend():
|
||||
"""Reset the singleton (for testing)."""
|
||||
global _backend
|
||||
_backend = None
|
||||
263
agent/memory/evaluation.py
Normal file
263
agent/memory/evaluation.py
Normal file
@@ -0,0 +1,263 @@
|
||||
"""Memory Backend Evaluation Framework.
|
||||
|
||||
Provides structured evaluation for comparing memory backends on:
|
||||
1. Latency (store/retrieve/query operations)
|
||||
2. Relevance (does query return useful results?)
|
||||
3. Privacy (where is data stored?)
|
||||
4. Reliability (availability, error handling)
|
||||
5. Cost (API calls, cloud dependency)
|
||||
|
||||
Usage:
|
||||
from agent.memory.evaluation import evaluate_backends
|
||||
report = evaluate_backends()
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
from dataclasses import dataclass, field, asdict
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class BackendEvaluation:
|
||||
"""Evaluation results for a single backend."""
|
||||
backend_name: str
|
||||
is_cloud: bool
|
||||
available: bool
|
||||
|
||||
# Latency (milliseconds)
|
||||
store_latency_ms: float = 0
|
||||
retrieve_latency_ms: float = 0
|
||||
query_latency_ms: float = 0
|
||||
|
||||
# Functionality
|
||||
store_success: bool = False
|
||||
retrieve_success: bool = False
|
||||
query_returns_results: bool = False
|
||||
query_result_count: int = 0
|
||||
|
||||
# Privacy
|
||||
data_location: str = "unknown"
|
||||
requires_api_key: bool = False
|
||||
|
||||
# Overall
|
||||
score: float = 0 # 0-100
|
||||
recommendation: str = ""
|
||||
notes: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
def _measure_latency(func, *args, **kwargs) -> tuple:
|
||||
"""Measure function latency in milliseconds."""
|
||||
start = time.perf_counter()
|
||||
try:
|
||||
result = func(*args, **kwargs)
|
||||
elapsed = (time.perf_counter() - start) * 1000
|
||||
return elapsed, result, None
|
||||
except Exception as e:
|
||||
elapsed = (time.perf_counter() - start) * 1000
|
||||
return elapsed, None, e
|
||||
|
||||
|
||||
def evaluate_backend(backend, test_user: str = "eval_user") -> BackendEvaluation:
|
||||
"""Evaluate a single memory backend."""
|
||||
from agent.memory import MemoryBackend
|
||||
|
||||
eval_result = BackendEvaluation(
|
||||
backend_name=backend.backend_name,
|
||||
is_cloud=backend.is_cloud,
|
||||
available=backend.is_available(),
|
||||
)
|
||||
|
||||
if not eval_result.available:
|
||||
eval_result.notes.append("Backend not available")
|
||||
eval_result.score = 0
|
||||
eval_result.recommendation = "NOT AVAILABLE"
|
||||
return eval_result
|
||||
|
||||
# Privacy assessment
|
||||
if backend.is_cloud:
|
||||
eval_result.data_location = "cloud (external)"
|
||||
eval_result.requires_api_key = True
|
||||
else:
|
||||
eval_result.data_location = "local (~/.hermes/)"
|
||||
|
||||
# Test store
|
||||
latency, success, err = _measure_latency(
|
||||
backend.store,
|
||||
test_user,
|
||||
"eval_test_key",
|
||||
"eval_test_value",
|
||||
{"source": "evaluation"},
|
||||
)
|
||||
eval_result.store_latency_ms = latency
|
||||
eval_result.store_success = success is True
|
||||
if err:
|
||||
eval_result.notes.append(f"Store error: {err}")
|
||||
|
||||
# Test retrieve
|
||||
latency, result, err = _measure_latency(
|
||||
backend.retrieve,
|
||||
test_user,
|
||||
"eval_test_key",
|
||||
)
|
||||
eval_result.retrieve_latency_ms = latency
|
||||
eval_result.retrieve_success = result is not None
|
||||
if err:
|
||||
eval_result.notes.append(f"Retrieve error: {err}")
|
||||
|
||||
# Test query
|
||||
latency, results, err = _measure_latency(
|
||||
backend.query,
|
||||
test_user,
|
||||
"eval_test",
|
||||
5,
|
||||
)
|
||||
eval_result.query_latency_ms = latency
|
||||
eval_result.query_returns_results = bool(results)
|
||||
eval_result.query_result_count = len(results) if results else 0
|
||||
if err:
|
||||
eval_result.notes.append(f"Query error: {err}")
|
||||
|
||||
# Cleanup
|
||||
try:
|
||||
backend.delete(test_user, "eval_test_key")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Score calculation (0-100)
|
||||
score = 0
|
||||
|
||||
# Availability (20 points)
|
||||
score += 20
|
||||
|
||||
# Functionality (40 points)
|
||||
if eval_result.store_success:
|
||||
score += 15
|
||||
if eval_result.retrieve_success:
|
||||
score += 15
|
||||
if eval_result.query_returns_results:
|
||||
score += 10
|
||||
|
||||
# Latency (20 points) — lower is better
|
||||
avg_latency = (
|
||||
eval_result.store_latency_ms +
|
||||
eval_result.retrieve_latency_ms +
|
||||
eval_result.query_latency_ms
|
||||
) / 3
|
||||
if avg_latency < 10:
|
||||
score += 20
|
||||
elif avg_latency < 50:
|
||||
score += 15
|
||||
elif avg_latency < 200:
|
||||
score += 10
|
||||
else:
|
||||
score += 5
|
||||
|
||||
# Privacy (20 points) — local is better for sovereignty
|
||||
if not backend.is_cloud:
|
||||
score += 20
|
||||
else:
|
||||
score += 5 # cloud has privacy trade-offs
|
||||
|
||||
eval_result.score = score
|
||||
|
||||
# Recommendation
|
||||
if score >= 80:
|
||||
eval_result.recommendation = "RECOMMENDED"
|
||||
elif score >= 60:
|
||||
eval_result.recommendation = "ACCEPTABLE"
|
||||
elif score >= 40:
|
||||
eval_result.recommendation = "MARGINAL"
|
||||
else:
|
||||
eval_result.recommendation = "NOT RECOMMENDED"
|
||||
|
||||
return eval_result
|
||||
|
||||
|
||||
def evaluate_backends() -> Dict[str, Any]:
|
||||
"""Evaluate all available memory backends.
|
||||
|
||||
Returns a comparison report.
|
||||
"""
|
||||
from agent.memory import NullBackend
|
||||
from agent.memory.local_backend import LocalBackend
|
||||
|
||||
backends = []
|
||||
|
||||
# Always evaluate Null (baseline)
|
||||
backends.append(NullBackend())
|
||||
|
||||
# Evaluate Local
|
||||
try:
|
||||
backends.append(LocalBackend())
|
||||
except Exception as e:
|
||||
logger.warning("Local backend init failed: %s", e)
|
||||
|
||||
# Try Honcho if configured
|
||||
import os
|
||||
if os.getenv("HONCHO_API_KEY"):
|
||||
try:
|
||||
from agent.memory.honcho_backend import HonchoBackend
|
||||
backends.append(HonchoBackend())
|
||||
except ImportError:
|
||||
logger.debug("Honcho not installed, skipping evaluation")
|
||||
|
||||
evaluations = []
|
||||
for backend in backends:
|
||||
try:
|
||||
evaluations.append(evaluate_backend(backend))
|
||||
except Exception as e:
|
||||
logger.warning("Evaluation failed for %s: %s", backend.backend_name, e)
|
||||
|
||||
# Build report
|
||||
report = {
|
||||
"timestamp": time.time(),
|
||||
"backends_evaluated": len(evaluations),
|
||||
"evaluations": [asdict(e) for e in evaluations],
|
||||
"recommendation": _build_recommendation(evaluations),
|
||||
}
|
||||
|
||||
return report
|
||||
|
||||
|
||||
def _build_recommendation(evaluations: List[BackendEvaluation]) -> str:
|
||||
"""Build overall recommendation from evaluations."""
|
||||
if not evaluations:
|
||||
return "No backends evaluated"
|
||||
|
||||
# Find best non-null backend
|
||||
viable = [e for e in evaluations if e.backend_name != "null (disabled)" and e.available]
|
||||
if not viable:
|
||||
return "No viable backends found. Use NullBackend (default)."
|
||||
|
||||
best = max(viable, key=lambda e: e.score)
|
||||
|
||||
parts = [f"Best backend: {best.backend_name} (score: {best.score})"]
|
||||
|
||||
if best.is_cloud:
|
||||
parts.append(
|
||||
"WARNING: Cloud backend has privacy trade-offs. "
|
||||
"Data leaves your machine. Consider LocalBackend for sovereignty."
|
||||
)
|
||||
|
||||
# Compare local vs cloud if both available
|
||||
local = [e for e in viable if not e.is_cloud]
|
||||
cloud = [e for e in viable if e.is_cloud]
|
||||
if local and cloud:
|
||||
local_score = max(e.score for e in local)
|
||||
cloud_score = max(e.score for e in cloud)
|
||||
if local_score >= cloud_score:
|
||||
parts.append(
|
||||
f"Local backend (score {local_score}) matches or beats "
|
||||
f"cloud (score {cloud_score}). RECOMMEND: stay local for sovereignty."
|
||||
)
|
||||
else:
|
||||
parts.append(
|
||||
f"Cloud backend (score {cloud_score}) outperforms "
|
||||
f"local (score {local_score}) but adds cloud dependency."
|
||||
)
|
||||
|
||||
return " ".join(parts)
|
||||
171
agent/memory/honcho_backend.py
Normal file
171
agent/memory/honcho_backend.py
Normal file
@@ -0,0 +1,171 @@
|
||||
"""Honcho memory backend — opt-in cloud-based user modeling.
|
||||
|
||||
Requires:
|
||||
- pip install honcho-ai
|
||||
- HONCHO_API_KEY environment variable (from app.honcho.dev)
|
||||
|
||||
Provides dialectic user context queries via Honcho's AI-native memory.
|
||||
Zero runtime overhead when not configured — get_memory_backend() falls
|
||||
back to LocalBackend or NullBackend if this fails to initialize.
|
||||
|
||||
This is the evaluation wrapper. It adapts the Honcho SDK to our
|
||||
MemoryBackend interface so we can A/B test against LocalBackend.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from agent.memory import MemoryBackend, MemoryEntry
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class HonchoBackend(MemoryBackend):
|
||||
"""Honcho AI-native memory backend.
|
||||
|
||||
Wraps the honcho-ai SDK to provide cross-session user modeling
|
||||
with dialectic context queries.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
self._client = None
|
||||
self._api_key = os.getenv("HONCHO_API_KEY", "")
|
||||
self._app_id = os.getenv("HONCHO_APP_ID", "hermes-agent")
|
||||
self._base_url = os.getenv("HONCHO_BASE_URL", "https://api.honcho.dev")
|
||||
|
||||
def _get_client(self):
|
||||
"""Lazy-load Honcho client."""
|
||||
if self._client is not None:
|
||||
return self._client
|
||||
|
||||
if not self._api_key:
|
||||
return None
|
||||
|
||||
try:
|
||||
from honcho import Honcho
|
||||
self._client = Honcho(
|
||||
api_key=self._api_key,
|
||||
app_id=self._app_id,
|
||||
base_url=self._base_url,
|
||||
)
|
||||
return self._client
|
||||
except ImportError:
|
||||
logger.warning("honcho-ai not installed. Install with: pip install honcho-ai")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.warning("Failed to initialize Honcho client: %s", e)
|
||||
return None
|
||||
|
||||
def is_available(self) -> bool:
|
||||
if not self._api_key:
|
||||
return False
|
||||
client = self._get_client()
|
||||
if client is None:
|
||||
return False
|
||||
# Try a simple API call to verify connectivity
|
||||
try:
|
||||
# Honcho uses sessions — verify we can list them
|
||||
client.get_sessions(limit=1)
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.debug("Honcho not available: %s", e)
|
||||
return False
|
||||
|
||||
def store(self, user_id: str, key: str, value: str, metadata: Dict = None) -> bool:
|
||||
client = self._get_client()
|
||||
if client is None:
|
||||
return False
|
||||
|
||||
try:
|
||||
# Honcho stores messages in sessions
|
||||
# We create a synthetic message to store the preference
|
||||
session_id = f"hermes-prefs-{user_id}"
|
||||
message_content = json.dumps({
|
||||
"type": "preference",
|
||||
"key": key,
|
||||
"value": value,
|
||||
"metadata": metadata or {},
|
||||
"timestamp": time.time(),
|
||||
})
|
||||
client.add_message(
|
||||
session_id=session_id,
|
||||
role="system",
|
||||
content=message_content,
|
||||
)
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.warning("Honcho store failed: %s", e)
|
||||
return False
|
||||
|
||||
def retrieve(self, user_id: str, key: str) -> Optional[MemoryEntry]:
|
||||
# Honcho doesn't have direct key-value retrieval
|
||||
# We query for the key and return the latest match
|
||||
results = self.query(user_id, key, limit=1)
|
||||
for entry in results:
|
||||
if entry.key == key:
|
||||
return entry
|
||||
return None
|
||||
|
||||
def query(self, user_id: str, query_text: str, limit: int = 10) -> List[MemoryEntry]:
|
||||
client = self._get_client()
|
||||
if client is None:
|
||||
return []
|
||||
|
||||
try:
|
||||
session_id = f"hermes-prefs-{user_id}"
|
||||
# Use Honcho's dialectic query
|
||||
result = client.chat(
|
||||
session_id=session_id,
|
||||
message=f"Find preferences related to: {query_text}",
|
||||
)
|
||||
|
||||
# Parse the response into memory entries
|
||||
entries = []
|
||||
if isinstance(result, dict):
|
||||
content = result.get("content", "")
|
||||
try:
|
||||
data = json.loads(content)
|
||||
if isinstance(data, list):
|
||||
for item in data[:limit]:
|
||||
entries.append(MemoryEntry(
|
||||
key=item.get("key", ""),
|
||||
value=item.get("value", ""),
|
||||
user_id=user_id,
|
||||
metadata=item.get("metadata", {}),
|
||||
))
|
||||
elif isinstance(data, dict) and data.get("key"):
|
||||
entries.append(MemoryEntry(
|
||||
key=data.get("key", ""),
|
||||
value=data.get("value", ""),
|
||||
user_id=user_id,
|
||||
metadata=data.get("metadata", {}),
|
||||
))
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
|
||||
return entries
|
||||
except Exception as e:
|
||||
logger.warning("Honcho query failed: %s", e)
|
||||
return []
|
||||
|
||||
def list_keys(self, user_id: str) -> List[str]:
|
||||
# Query all and extract keys
|
||||
results = self.query(user_id, "", limit=100)
|
||||
return list(dict.fromkeys(e.key for e in results if e.key))
|
||||
|
||||
def delete(self, user_id: str, key: str) -> bool:
|
||||
# Honcho doesn't support deletion of individual entries
|
||||
# This is a limitation of the cloud backend
|
||||
logger.info("Honcho does not support individual entry deletion")
|
||||
return False
|
||||
|
||||
@property
|
||||
def backend_name(self) -> str:
|
||||
return "honcho (cloud)"
|
||||
|
||||
@property
|
||||
def is_cloud(self) -> bool:
|
||||
return True
|
||||
156
agent/memory/local_backend.py
Normal file
156
agent/memory/local_backend.py
Normal file
@@ -0,0 +1,156 @@
|
||||
"""Local SQLite memory backend.
|
||||
|
||||
Zero cloud dependency. Stores user preferences and patterns in a
|
||||
local SQLite database at ~/.hermes/memory.db.
|
||||
|
||||
Provides basic key-value storage with simple text search.
|
||||
No external dependencies beyond Python stdlib.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import sqlite3
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from hermes_constants import get_hermes_home
|
||||
from agent.memory import MemoryBackend, MemoryEntry
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class LocalBackend(MemoryBackend):
|
||||
"""SQLite-backed local memory storage."""
|
||||
|
||||
def __init__(self, db_path: Path = None):
|
||||
self._db_path = db_path or (get_hermes_home() / "memory.db")
|
||||
self._init_db()
|
||||
|
||||
def _init_db(self):
|
||||
"""Initialize the database schema."""
|
||||
self._db_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with sqlite3.connect(str(self._db_path)) as conn:
|
||||
conn.execute("""
|
||||
CREATE TABLE IF NOT EXISTS memories (
|
||||
user_id TEXT NOT NULL,
|
||||
key TEXT NOT NULL,
|
||||
value TEXT NOT NULL,
|
||||
metadata TEXT,
|
||||
created_at REAL NOT NULL,
|
||||
updated_at REAL NOT NULL,
|
||||
PRIMARY KEY (user_id, key)
|
||||
)
|
||||
""")
|
||||
conn.execute("""
|
||||
CREATE INDEX IF NOT EXISTS idx_memories_user
|
||||
ON memories(user_id)
|
||||
""")
|
||||
conn.commit()
|
||||
|
||||
def is_available(self) -> bool:
|
||||
try:
|
||||
with sqlite3.connect(str(self._db_path)) as conn:
|
||||
conn.execute("SELECT 1")
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def store(self, user_id: str, key: str, value: str, metadata: Dict = None) -> bool:
|
||||
try:
|
||||
now = time.time()
|
||||
meta_json = json.dumps(metadata) if metadata else None
|
||||
with sqlite3.connect(str(self._db_path)) as conn:
|
||||
conn.execute("""
|
||||
INSERT INTO memories (user_id, key, value, metadata, created_at, updated_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(user_id, key) DO UPDATE SET
|
||||
value = excluded.value,
|
||||
metadata = excluded.metadata,
|
||||
updated_at = excluded.updated_at
|
||||
""", (user_id, key, value, meta_json, now, now))
|
||||
conn.commit()
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.warning("Failed to store memory: %s", e)
|
||||
return False
|
||||
|
||||
def retrieve(self, user_id: str, key: str) -> Optional[MemoryEntry]:
|
||||
try:
|
||||
with sqlite3.connect(str(self._db_path)) as conn:
|
||||
row = conn.execute(
|
||||
"SELECT key, value, user_id, created_at, updated_at, metadata "
|
||||
"FROM memories WHERE user_id = ? AND key = ?",
|
||||
(user_id, key),
|
||||
).fetchone()
|
||||
if not row:
|
||||
return None
|
||||
return MemoryEntry(
|
||||
key=row[0],
|
||||
value=row[1],
|
||||
user_id=row[2],
|
||||
created_at=row[3],
|
||||
updated_at=row[4],
|
||||
metadata=json.loads(row[5]) if row[5] else {},
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning("Failed to retrieve memory: %s", e)
|
||||
return None
|
||||
|
||||
def query(self, user_id: str, query_text: str, limit: int = 10) -> List[MemoryEntry]:
|
||||
"""Simple LIKE-based search on keys and values."""
|
||||
try:
|
||||
pattern = f"%{query_text}%"
|
||||
with sqlite3.connect(str(self._db_path)) as conn:
|
||||
rows = conn.execute("""
|
||||
SELECT key, value, user_id, created_at, updated_at, metadata
|
||||
FROM memories
|
||||
WHERE user_id = ? AND (key LIKE ? OR value LIKE ?)
|
||||
ORDER BY updated_at DESC
|
||||
LIMIT ?
|
||||
""", (user_id, pattern, pattern, limit)).fetchall()
|
||||
return [
|
||||
MemoryEntry(
|
||||
key=r[0],
|
||||
value=r[1],
|
||||
user_id=r[2],
|
||||
created_at=r[3],
|
||||
updated_at=r[4],
|
||||
metadata=json.loads(r[5]) if r[5] else {},
|
||||
)
|
||||
for r in rows
|
||||
]
|
||||
except Exception as e:
|
||||
logger.warning("Failed to query memories: %s", e)
|
||||
return []
|
||||
|
||||
def list_keys(self, user_id: str) -> List[str]:
|
||||
try:
|
||||
with sqlite3.connect(str(self._db_path)) as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT key FROM memories WHERE user_id = ? ORDER BY updated_at DESC",
|
||||
(user_id,),
|
||||
).fetchall()
|
||||
return [r[0] for r in rows]
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
def delete(self, user_id: str, key: str) -> bool:
|
||||
try:
|
||||
with sqlite3.connect(str(self._db_path)) as conn:
|
||||
conn.execute(
|
||||
"DELETE FROM memories WHERE user_id = ? AND key = ?",
|
||||
(user_id, key),
|
||||
)
|
||||
conn.commit()
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
@property
|
||||
def backend_name(self) -> str:
|
||||
return "local (SQLite)"
|
||||
|
||||
@property
|
||||
def is_cloud(self) -> bool:
|
||||
return False
|
||||
@@ -5258,80 +5258,6 @@ For more help on a command:
|
||||
|
||||
sessions_parser.set_defaults(func=cmd_sessions)
|
||||
|
||||
|
||||
# Warm session command
|
||||
warm_parser = subparsers.add_parser(
|
||||
"warm",
|
||||
help="Warm session provisioning",
|
||||
description="Create pre-contextualized sessions from templates"
|
||||
)
|
||||
warm_subparsers = warm_parser.add_subparsers(dest="warm_command")
|
||||
|
||||
# Extract command
|
||||
warm_extract = warm_subparsers.add_parser("extract", help="Extract template from session")
|
||||
warm_extract.add_argument("session_id", help="Session ID to extract from")
|
||||
warm_extract.add_argument("--name", "-n", required=True, help="Template name")
|
||||
warm_extract.add_argument("--description", "-d", default="", help="Template description")
|
||||
|
||||
# List command
|
||||
warm_subparsers.add_parser("list", help="List available templates")
|
||||
|
||||
# Test command
|
||||
warm_test = warm_subparsers.add_parser("test", help="Test warm session creation")
|
||||
warm_test.add_argument("template_id", help="Template ID")
|
||||
warm_test.add_argument("message", help="Test message")
|
||||
|
||||
# Delete command
|
||||
warm_delete = warm_subparsers.add_parser("delete", help="Delete a template")
|
||||
warm_delete.add_argument("template_id", help="Template ID to delete")
|
||||
|
||||
warm_parser.set_defaults(func=cmd_warm)
|
||||
|
||||
# A/B testing command
|
||||
ab_parser = subparsers.add_parser(
|
||||
"ab-test",
|
||||
help="A/B test warm vs cold sessions",
|
||||
description="Framework for comparing warm and cold session performance"
|
||||
)
|
||||
ab_subparsers = ab_parser.add_subparsers(dest="ab_command")
|
||||
|
||||
# Create test
|
||||
ab_create = ab_subparsers.add_parser("create", help="Create a new A/B test")
|
||||
ab_create.add_argument("--task-id", required=True, help="Task ID")
|
||||
ab_create.add_argument("--description", required=True, help="Task description")
|
||||
ab_create.add_argument("--prompt", required=True, help="Test prompt")
|
||||
ab_create.add_argument("--category", default="general", help="Task category")
|
||||
ab_create.add_argument("--difficulty", default="medium", choices=["easy", "medium", "hard"])
|
||||
|
||||
# List tests
|
||||
ab_subparsers.add_parser("list", help="List all A/B tests")
|
||||
|
||||
# Show test
|
||||
ab_show = ab_subparsers.add_parser("show", help="Show test details")
|
||||
ab_show.add_argument("test_id", help="Test ID")
|
||||
|
||||
# Analyze test
|
||||
ab_analyze = ab_subparsers.add_parser("analyze", help="Analyze test results")
|
||||
ab_analyze.add_argument("test_id", help="Test ID")
|
||||
|
||||
# Add result
|
||||
ab_add = ab_subparsers.add_parser("add-result", help="Add a test result")
|
||||
ab_add.add_argument("test_id", help="Test ID")
|
||||
ab_add.add_argument("--session-type", required=True, choices=["cold", "warm"])
|
||||
ab_add.add_argument("--session-id", required=True, help="Session ID")
|
||||
ab_add.add_argument("--tool-calls", type=int, default=0)
|
||||
ab_add.add_argument("--successful-calls", type=int, default=0)
|
||||
ab_add.add_argument("--completion-time", type=float, default=0.0)
|
||||
ab_add.add_argument("--success", action="store_true")
|
||||
ab_add.add_argument("--notes", default="")
|
||||
|
||||
# Delete test
|
||||
ab_delete = ab_subparsers.add_parser("delete", help="Delete a test")
|
||||
ab_delete.add_argument("test_id", help="Test ID")
|
||||
|
||||
ab_parser.set_defaults(func=cmd_ab_test)
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# insights command
|
||||
# =========================================================================
|
||||
@@ -5672,102 +5598,3 @@ Examples:
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
|
||||
def cmd_warm(args):
|
||||
"""Handle warm session commands."""
|
||||
from hermes_cli.colors import Colors, color
|
||||
|
||||
subcmd = getattr(args, 'warm_command', None)
|
||||
|
||||
if subcmd is None:
|
||||
print(color("Warm Session Provisioning", Colors.CYAN))
|
||||
print("\nCommands:")
|
||||
print(" hermes warm extract SESSION_ID --name NAME - Extract template from session")
|
||||
print(" hermes warm list - List available templates")
|
||||
print(" hermes warm test TEMPLATE_ID MESSAGE - Test warm session")
|
||||
print(" hermes warm delete TEMPLATE_ID - Delete a template")
|
||||
return 0
|
||||
|
||||
try:
|
||||
from tools.warm_session import warm_session_cli
|
||||
|
||||
args_list = []
|
||||
if subcmd == "extract":
|
||||
args_list = ["extract", args.session_id, "--name", args.name]
|
||||
if args.description:
|
||||
args_list.extend(["--description", args.description])
|
||||
elif subcmd == "list":
|
||||
args_list = ["list"]
|
||||
elif subcmd == "test":
|
||||
args_list = ["test", args.template_id, args.message]
|
||||
elif subcmd == "delete":
|
||||
args_list = ["delete", args.template_id]
|
||||
|
||||
return warm_session_cli(args_list)
|
||||
|
||||
except ImportError as e:
|
||||
print(color(f"Error: Cannot import warm_session module: {e}", Colors.RED))
|
||||
return 1
|
||||
except Exception as e:
|
||||
print(color(f"Error: {e}", Colors.RED))
|
||||
return 1
|
||||
|
||||
|
||||
def cmd_ab_test(args):
|
||||
"""Handle A/B testing commands."""
|
||||
from hermes_cli.colors import Colors, color
|
||||
|
||||
subcmd = getattr(args, 'ab_command', None)
|
||||
|
||||
if subcmd is None:
|
||||
print(color("A/B Testing Framework for Warm vs Cold Sessions", Colors.CYAN))
|
||||
print("\nCommands:")
|
||||
print(" hermes ab-test create --task-id ID --description DESC --prompt PROMPT")
|
||||
print(" hermes ab-test list")
|
||||
print(" hermes ab-test show TEST_ID")
|
||||
print(" hermes ab-test analyze TEST_ID")
|
||||
print(" hermes ab-test add-result TEST_ID --session-type TYPE --session-id ID")
|
||||
print(" hermes ab-test delete TEST_ID")
|
||||
return 0
|
||||
|
||||
try:
|
||||
from tools.session_ab_testing import ab_test_cli
|
||||
|
||||
args_list = []
|
||||
if subcmd == "create":
|
||||
args_list = ["create", "--task-id", args.task_id, "--description", args.description, "--prompt", args.prompt]
|
||||
if args.category:
|
||||
args_list.extend(["--category", args.category])
|
||||
if args.difficulty:
|
||||
args_list.extend(["--difficulty", args.difficulty])
|
||||
elif subcmd == "list":
|
||||
args_list = ["list"]
|
||||
elif subcmd == "show":
|
||||
args_list = ["show", args.test_id]
|
||||
elif subcmd == "analyze":
|
||||
args_list = ["analyze", args.test_id]
|
||||
elif subcmd == "add-result":
|
||||
args_list = ["add-result", args.test_id, "--session-type", args.session_type, "--session-id", args.session_id]
|
||||
if args.tool_calls:
|
||||
args_list.extend(["--tool-calls", str(args.tool_calls)])
|
||||
if args.successful_calls:
|
||||
args_list.extend(["--successful-calls", str(args.successful_calls)])
|
||||
if args.completion_time:
|
||||
args_list.extend(["--completion-time", str(args.completion_time)])
|
||||
if args.success:
|
||||
args_list.append("--success")
|
||||
if args.notes:
|
||||
args_list.extend(["--notes", args.notes])
|
||||
elif subcmd == "delete":
|
||||
args_list = ["delete", args.test_id]
|
||||
|
||||
return ab_test_cli(args_list)
|
||||
|
||||
except ImportError as e:
|
||||
print(color(f"Error: Cannot import session_ab_testing module: {e}", Colors.RED))
|
||||
return 1
|
||||
except Exception as e:
|
||||
print(color(f"Error: {e}", Colors.RED))
|
||||
return 1
|
||||
|
||||
|
||||
28
run_agent.py
28
run_agent.py
@@ -1001,10 +1001,30 @@ class AIAgent:
|
||||
self._session_db = session_db
|
||||
self._parent_session_id = parent_session_id
|
||||
self._last_flushed_db_idx = 0 # tracks DB-write cursor to prevent duplicate writes
|
||||
# Lazy session creation: defer until first message flush (#314).
|
||||
# _flush_messages_to_session_db() calls ensure_session() which uses
|
||||
# INSERT OR IGNORE — creating the row only when messages arrive.
|
||||
# This eliminates 32% of sessions that are created but never used.
|
||||
if self._session_db:
|
||||
try:
|
||||
self._session_db.create_session(
|
||||
session_id=self.session_id,
|
||||
source=self.platform or os.environ.get("HERMES_SESSION_SOURCE", "cli"),
|
||||
model=self.model,
|
||||
model_config={
|
||||
"max_iterations": self.max_iterations,
|
||||
"reasoning_config": reasoning_config,
|
||||
"max_tokens": max_tokens,
|
||||
},
|
||||
user_id=None,
|
||||
parent_session_id=self._parent_session_id,
|
||||
)
|
||||
except Exception as e:
|
||||
# Transient SQLite lock contention (e.g. CLI and gateway writing
|
||||
# concurrently) must NOT permanently disable session_search for
|
||||
# this agent. Keep _session_db alive — subsequent message
|
||||
# flushes and session_search calls will still work once the
|
||||
# lock clears. The session row may be missing from the index
|
||||
# for this run, but that is recoverable (flushes upsert rows).
|
||||
logger.warning(
|
||||
"Session DB create_session failed (session_search still available): %s", e
|
||||
)
|
||||
|
||||
# In-memory todo list for task planning (one per agent/session)
|
||||
from tools.todo_tool import TodoStore
|
||||
|
||||
205
tests/agent/test_memory_backend.py
Normal file
205
tests/agent/test_memory_backend.py
Normal file
@@ -0,0 +1,205 @@
|
||||
"""Tests for memory backend system (#322)."""
|
||||
|
||||
import json
|
||||
import time
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from agent.memory import (
|
||||
MemoryEntry,
|
||||
NullBackend,
|
||||
get_memory_backend,
|
||||
reset_backend,
|
||||
)
|
||||
from agent.memory.local_backend import LocalBackend
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def isolated_local_backend(tmp_path, monkeypatch):
|
||||
"""Create a LocalBackend with temp DB."""
|
||||
db_path = tmp_path / "test_memory.db"
|
||||
return LocalBackend(db_path=db_path)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def reset_memory():
|
||||
"""Reset the memory backend singleton."""
|
||||
reset_backend()
|
||||
yield
|
||||
reset_backend()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# MemoryEntry
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestMemoryEntry:
|
||||
def test_creation(self):
|
||||
entry = MemoryEntry(key="pref", value="python", user_id="u1")
|
||||
assert entry.key == "pref"
|
||||
assert entry.value == "python"
|
||||
assert entry.created_at > 0
|
||||
|
||||
def test_defaults(self):
|
||||
entry = MemoryEntry(key="k", value="v", user_id="u1")
|
||||
assert entry.metadata == {}
|
||||
assert entry.updated_at == entry.created_at
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# NullBackend
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestNullBackend:
|
||||
def test_always_available(self):
|
||||
backend = NullBackend()
|
||||
assert backend.is_available() is True
|
||||
|
||||
def test_store_noop(self):
|
||||
backend = NullBackend()
|
||||
assert backend.store("u1", "k", "v") is True
|
||||
|
||||
def test_retrieve_returns_none(self):
|
||||
backend = NullBackend()
|
||||
assert backend.retrieve("u1", "k") is None
|
||||
|
||||
def test_query_returns_empty(self):
|
||||
backend = NullBackend()
|
||||
assert backend.query("u1", "test") == []
|
||||
|
||||
def test_not_cloud(self):
|
||||
backend = NullBackend()
|
||||
assert backend.is_cloud is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# LocalBackend
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestLocalBackend:
|
||||
def test_available(self, isolated_local_backend):
|
||||
assert isolated_local_backend.is_available() is True
|
||||
|
||||
def test_store_and_retrieve(self, isolated_local_backend):
|
||||
assert isolated_local_backend.store("u1", "lang", "python")
|
||||
entry = isolated_local_backend.retrieve("u1", "lang")
|
||||
assert entry is not None
|
||||
assert entry.value == "python"
|
||||
assert entry.key == "lang"
|
||||
|
||||
def test_store_with_metadata(self, isolated_local_backend):
|
||||
assert isolated_local_backend.store("u1", "k", "v", {"source": "test"})
|
||||
entry = isolated_local_backend.retrieve("u1", "k")
|
||||
assert entry.metadata == {"source": "test"}
|
||||
|
||||
def test_update_existing(self, isolated_local_backend):
|
||||
isolated_local_backend.store("u1", "k", "v1")
|
||||
isolated_local_backend.store("u1", "k", "v2")
|
||||
entry = isolated_local_backend.retrieve("u1", "k")
|
||||
assert entry.value == "v2"
|
||||
|
||||
def test_query(self, isolated_local_backend):
|
||||
isolated_local_backend.store("u1", "pref_python", "True")
|
||||
isolated_local_backend.store("u1", "pref_editor", "vim")
|
||||
isolated_local_backend.store("u1", "theme", "dark")
|
||||
|
||||
results = isolated_local_backend.query("u1", "pref")
|
||||
assert len(results) == 2
|
||||
keys = {r.key for r in results}
|
||||
assert "pref_python" in keys
|
||||
assert "pref_editor" in keys
|
||||
|
||||
def test_list_keys(self, isolated_local_backend):
|
||||
isolated_local_backend.store("u1", "a", "1")
|
||||
isolated_local_backend.store("u1", "b", "2")
|
||||
keys = isolated_local_backend.list_keys("u1")
|
||||
assert set(keys) == {"a", "b"}
|
||||
|
||||
def test_delete(self, isolated_local_backend):
|
||||
isolated_local_backend.store("u1", "k", "v")
|
||||
assert isolated_local_backend.delete("u1", "k")
|
||||
assert isolated_local_backend.retrieve("u1", "k") is None
|
||||
|
||||
def test_retrieve_nonexistent(self, isolated_local_backend):
|
||||
assert isolated_local_backend.retrieve("u1", "nope") is None
|
||||
|
||||
def test_not_cloud(self, isolated_local_backend):
|
||||
assert isolated_local_backend.is_cloud is False
|
||||
|
||||
def test_separate_users(self, isolated_local_backend):
|
||||
isolated_local_backend.store("u1", "k", "user1_value")
|
||||
isolated_local_backend.store("u2", "k", "user2_value")
|
||||
assert isolated_local_backend.retrieve("u1", "k").value == "user1_value"
|
||||
assert isolated_local_backend.retrieve("u2", "k").value == "user2_value"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Singleton
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestSingleton:
|
||||
def test_default_is_null(self, reset_memory, monkeypatch):
|
||||
monkeypatch.delenv("HERMES_MEMORY_BACKEND", raising=False)
|
||||
monkeypatch.delenv("HONCHO_API_KEY", raising=False)
|
||||
backend = get_memory_backend()
|
||||
assert isinstance(backend, NullBackend)
|
||||
|
||||
def test_local_when_configured(self, reset_memory, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_MEMORY_BACKEND", "local")
|
||||
backend = get_memory_backend()
|
||||
assert isinstance(backend, LocalBackend)
|
||||
|
||||
def test_caches_instance(self, reset_memory, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_MEMORY_BACKEND", "local")
|
||||
b1 = get_memory_backend()
|
||||
b2 = get_memory_backend()
|
||||
assert b1 is b2
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# HonchoBackend (mocked)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestHonchoBackend:
|
||||
def test_not_available_without_key(self, monkeypatch):
|
||||
monkeypatch.delenv("HONCHO_API_KEY", raising=False)
|
||||
from agent.memory.honcho_backend import HonchoBackend
|
||||
backend = HonchoBackend()
|
||||
assert backend.is_available() is False
|
||||
|
||||
def test_is_cloud(self):
|
||||
from agent.memory.honcho_backend import HonchoBackend
|
||||
backend = HonchoBackend()
|
||||
assert backend.is_cloud is True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Evaluation framework
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestEvaluation:
|
||||
def test_evaluate_null_backend(self):
|
||||
from agent.memory.evaluation import evaluate_backend
|
||||
result = evaluate_backend(NullBackend())
|
||||
assert result.backend_name == "null (disabled)"
|
||||
assert result.available is True
|
||||
assert result.score > 0
|
||||
assert result.is_cloud is False
|
||||
|
||||
def test_evaluate_local_backend(self, isolated_local_backend):
|
||||
from agent.memory.evaluation import evaluate_backend
|
||||
result = evaluate_backend(isolated_local_backend)
|
||||
assert result.backend_name == "local (SQLite)"
|
||||
assert result.available is True
|
||||
assert result.store_success is True
|
||||
assert result.retrieve_success is True
|
||||
assert result.score >= 80 # local should score well
|
||||
|
||||
def test_evaluate_backends_returns_report(self, reset_memory, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_MEMORY_BACKEND", "local")
|
||||
from agent.memory.evaluation import evaluate_backends
|
||||
report = evaluate_backends()
|
||||
assert "backends_evaluated" in report
|
||||
assert report["backends_evaluated"] >= 2 # null + local
|
||||
assert "recommendation" in report
|
||||
165
tools/memory_backend_tool.py
Normal file
165
tools/memory_backend_tool.py
Normal file
@@ -0,0 +1,165 @@
|
||||
"""Memory Backend Tool — manage cross-session memory backends.
|
||||
|
||||
Provides store/retrieve/query/evaluate/list actions for the
|
||||
pluggable memory backend system.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
from tools.registry import registry
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def memory_backend(
|
||||
action: str,
|
||||
user_id: str = "default",
|
||||
key: str = None,
|
||||
value: str = None,
|
||||
query_text: str = None,
|
||||
metadata: dict = None,
|
||||
) -> str:
|
||||
"""Manage cross-session memory backends.
|
||||
|
||||
Actions:
|
||||
store — store a user preference/pattern
|
||||
retrieve — retrieve a specific memory by key
|
||||
query — search memories by text
|
||||
list — list all keys for a user
|
||||
delete — delete a memory entry
|
||||
info — show current backend info
|
||||
evaluate — run evaluation framework comparing backends
|
||||
"""
|
||||
from agent.memory import get_memory_backend
|
||||
|
||||
backend = get_memory_backend()
|
||||
|
||||
if action == "info":
|
||||
return json.dumps({
|
||||
"success": True,
|
||||
"backend": backend.backend_name,
|
||||
"is_cloud": backend.is_cloud,
|
||||
"available": backend.is_available(),
|
||||
})
|
||||
|
||||
if action == "store":
|
||||
if not key or value is None:
|
||||
return json.dumps({"success": False, "error": "key and value are required for 'store'."})
|
||||
success = backend.store(user_id, key, value, metadata)
|
||||
return json.dumps({"success": success, "key": key})
|
||||
|
||||
if action == "retrieve":
|
||||
if not key:
|
||||
return json.dumps({"success": False, "error": "key is required for 'retrieve'."})
|
||||
entry = backend.retrieve(user_id, key)
|
||||
if entry is None:
|
||||
return json.dumps({"success": False, "error": f"No memory found for key '{key}'."})
|
||||
return json.dumps({
|
||||
"success": True,
|
||||
"key": entry.key,
|
||||
"value": entry.value,
|
||||
"metadata": entry.metadata,
|
||||
"updated_at": entry.updated_at,
|
||||
})
|
||||
|
||||
if action == "query":
|
||||
if not query_text:
|
||||
return json.dumps({"success": False, "error": "query_text is required for 'query'."})
|
||||
results = backend.query(user_id, query_text)
|
||||
return json.dumps({
|
||||
"success": True,
|
||||
"results": [
|
||||
{"key": e.key, "value": e.value, "metadata": e.metadata}
|
||||
for e in results
|
||||
],
|
||||
"count": len(results),
|
||||
})
|
||||
|
||||
if action == "list":
|
||||
keys = backend.list_keys(user_id)
|
||||
return json.dumps({"success": True, "keys": keys, "count": len(keys)})
|
||||
|
||||
if action == "delete":
|
||||
if not key:
|
||||
return json.dumps({"success": False, "error": "key is required for 'delete'."})
|
||||
success = backend.delete(user_id, key)
|
||||
return json.dumps({"success": success})
|
||||
|
||||
if action == "evaluate":
|
||||
from agent.memory.evaluation import evaluate_backends
|
||||
report = evaluate_backends()
|
||||
return json.dumps({
|
||||
"success": True,
|
||||
**report,
|
||||
})
|
||||
|
||||
return json.dumps({
|
||||
"success": False,
|
||||
"error": f"Unknown action '{action}'. Use: store, retrieve, query, list, delete, info, evaluate",
|
||||
})
|
||||
|
||||
|
||||
MEMORY_BACKEND_SCHEMA = {
|
||||
"name": "memory_backend",
|
||||
"description": (
|
||||
"Manage cross-session memory backends for user preference persistence. "
|
||||
"Pluggable architecture supports local SQLite (default, zero cloud dependency) "
|
||||
"and optional Honcho cloud backend (requires HONCHO_API_KEY).\n\n"
|
||||
"Actions:\n"
|
||||
" store — store a user preference/pattern\n"
|
||||
" retrieve — retrieve a specific memory by key\n"
|
||||
" query — search memories by text\n"
|
||||
" list — list all keys for a user\n"
|
||||
" delete — delete a memory entry\n"
|
||||
" info — show current backend info\n"
|
||||
" evaluate — run evaluation framework comparing backends"
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"action": {
|
||||
"type": "string",
|
||||
"enum": ["store", "retrieve", "query", "list", "delete", "info", "evaluate"],
|
||||
"description": "The action to perform.",
|
||||
},
|
||||
"user_id": {
|
||||
"type": "string",
|
||||
"description": "User identifier for memory operations (default: 'default').",
|
||||
},
|
||||
"key": {
|
||||
"type": "string",
|
||||
"description": "Memory key for store/retrieve/delete.",
|
||||
},
|
||||
"value": {
|
||||
"type": "string",
|
||||
"description": "Value to store.",
|
||||
},
|
||||
"query_text": {
|
||||
"type": "string",
|
||||
"description": "Search text for query action.",
|
||||
},
|
||||
"metadata": {
|
||||
"type": "object",
|
||||
"description": "Optional metadata dict for store.",
|
||||
},
|
||||
},
|
||||
"required": ["action"],
|
||||
},
|
||||
}
|
||||
|
||||
registry.register(
|
||||
name="memory_backend",
|
||||
toolset="skills",
|
||||
schema=MEMORY_BACKEND_SCHEMA,
|
||||
handler=lambda args, **kw: memory_backend(
|
||||
action=args.get("action", ""),
|
||||
user_id=args.get("user_id", "default"),
|
||||
key=args.get("key"),
|
||||
value=args.get("value"),
|
||||
query_text=args.get("query_text"),
|
||||
metadata=args.get("metadata"),
|
||||
),
|
||||
emoji="🧠",
|
||||
)
|
||||
@@ -1,517 +0,0 @@
|
||||
"""
|
||||
Warm Session A/B Testing Framework
|
||||
|
||||
Framework for comparing warm vs cold session performance.
|
||||
Addresses research questions from issue #327.
|
||||
|
||||
Issue: #327
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from dataclasses import dataclass, asdict, field
|
||||
from enum import Enum
|
||||
import statistics
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class SessionType(Enum):
|
||||
"""Type of session for A/B testing."""
|
||||
COLD = "cold" # Fresh session, no warm-up
|
||||
WARM = "warm" # Session with warm-up context
|
||||
|
||||
|
||||
@dataclass
|
||||
class TestTask:
|
||||
"""A task for A/B testing."""
|
||||
task_id: str
|
||||
description: str
|
||||
prompt: str
|
||||
expected_tools: List[str] = field(default_factory=list)
|
||||
success_criteria: Dict[str, Any] = field(default_factory=dict)
|
||||
category: str = "general"
|
||||
difficulty: str = "medium" # easy, medium, hard
|
||||
|
||||
|
||||
@dataclass
|
||||
class SessionResult:
|
||||
"""Result from a session test."""
|
||||
session_id: str
|
||||
session_type: SessionType
|
||||
task_id: str
|
||||
start_time: str
|
||||
end_time: Optional[str] = None
|
||||
message_count: int = 0
|
||||
tool_calls: int = 0
|
||||
successful_tool_calls: int = 0
|
||||
errors: List[str] = field(default_factory=list)
|
||||
completion_time_seconds: float = 0.0
|
||||
user_corrections: int = 0
|
||||
success: bool = False
|
||||
notes: str = ""
|
||||
|
||||
@property
|
||||
def error_rate(self) -> float:
|
||||
"""Calculate error rate."""
|
||||
if self.tool_calls == 0:
|
||||
return 0.0
|
||||
return (self.tool_calls - self.successful_tool_calls) / self.tool_calls
|
||||
|
||||
@property
|
||||
def success_rate(self) -> float:
|
||||
"""Calculate success rate."""
|
||||
if self.tool_calls == 0:
|
||||
return 0.0
|
||||
return self.successful_tool_calls / self.tool_calls
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {
|
||||
"session_id": self.session_id,
|
||||
"session_type": self.session_type.value,
|
||||
"task_id": self.task_id,
|
||||
"start_time": self.start_time,
|
||||
"end_time": self.end_time,
|
||||
"message_count": self.message_count,
|
||||
"tool_calls": self.tool_calls,
|
||||
"successful_tool_calls": self.successful_tool_calls,
|
||||
"errors": self.errors,
|
||||
"completion_time_seconds": self.completion_time_seconds,
|
||||
"user_corrections": self.user_corrections,
|
||||
"success": self.success,
|
||||
"error_rate": self.error_rate,
|
||||
"success_rate": self.success_rate,
|
||||
"notes": self.notes
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class ABTestResult:
|
||||
"""Results from an A/B test."""
|
||||
test_id: str
|
||||
task: TestTask
|
||||
cold_results: List[SessionResult] = field(default_factory=list)
|
||||
warm_results: List[SessionResult] = field(default_factory=list)
|
||||
created_at: str = field(default_factory=lambda: datetime.now().isoformat())
|
||||
|
||||
def add_result(self, result: SessionResult):
|
||||
"""Add a session result."""
|
||||
if result.session_type == SessionType.COLD:
|
||||
self.cold_results.append(result)
|
||||
else:
|
||||
self.warm_results.append(result)
|
||||
|
||||
def get_summary(self) -> Dict[str, Any]:
|
||||
"""Get summary statistics."""
|
||||
def calc_stats(results: List[SessionResult]) -> Dict[str, Any]:
|
||||
if not results:
|
||||
return {"count": 0}
|
||||
|
||||
error_rates = [r.error_rate for r in results]
|
||||
success_rates = [r.success_rate for r in results]
|
||||
completion_times = [r.completion_time_seconds for r in results if r.completion_time_seconds > 0]
|
||||
message_counts = [r.message_count for r in results]
|
||||
|
||||
return {
|
||||
"count": len(results),
|
||||
"avg_error_rate": statistics.mean(error_rates) if error_rates else 0,
|
||||
"avg_success_rate": statistics.mean(success_rates) if success_rates else 0,
|
||||
"avg_completion_time": statistics.mean(completion_times) if completion_times else 0,
|
||||
"avg_messages": statistics.mean(message_counts) if message_counts else 0,
|
||||
"success_count": sum(1 for r in results if r.success)
|
||||
}
|
||||
|
||||
cold_stats = calc_stats(self.cold_results)
|
||||
warm_stats = calc_stats(self.warm_results)
|
||||
|
||||
# Calculate improvement
|
||||
improvement = {}
|
||||
if cold_stats.get("count", 0) > 0 and warm_stats.get("count", 0) > 0:
|
||||
cold_error = cold_stats.get("avg_error_rate", 0)
|
||||
warm_error = warm_stats.get("avg_error_rate", 0)
|
||||
|
||||
if cold_error > 0:
|
||||
improvement["error_rate"] = (cold_error - warm_error) / cold_error
|
||||
|
||||
cold_success = cold_stats.get("avg_success_rate", 0)
|
||||
warm_success = warm_stats.get("avg_success_rate", 0)
|
||||
|
||||
if cold_success > 0:
|
||||
improvement["success_rate"] = (warm_success - cold_success) / cold_success
|
||||
|
||||
return {
|
||||
"task_id": self.task.task_id,
|
||||
"cold": cold_stats,
|
||||
"warm": warm_stats,
|
||||
"improvement": improvement,
|
||||
"recommendation": self._get_recommendation(cold_stats, warm_stats)
|
||||
}
|
||||
|
||||
def _get_recommendation(self, cold_stats: Dict, warm_stats: Dict) -> str:
|
||||
"""Generate recommendation based on results."""
|
||||
if cold_stats.get("count", 0) < 3 or warm_stats.get("count", 0) < 3:
|
||||
return "Insufficient data (need at least 3 tests each)"
|
||||
|
||||
cold_error = cold_stats.get("avg_error_rate", 0)
|
||||
warm_error = warm_stats.get("avg_error_rate", 0)
|
||||
|
||||
if warm_error < cold_error * 0.8: # 20% improvement
|
||||
return "WARM recommended: Significant error reduction"
|
||||
elif warm_error > cold_error * 1.2: # 20% worse
|
||||
return "COLD recommended: Warm sessions performed worse"
|
||||
else:
|
||||
return "No significant difference detected"
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {
|
||||
"test_id": self.test_id,
|
||||
"task": asdict(self.task),
|
||||
"cold_results": [r.to_dict() for r in self.cold_results],
|
||||
"warm_results": [r.to_dict() for r in self.warm_results],
|
||||
"created_at": self.created_at,
|
||||
"summary": self.get_summary()
|
||||
}
|
||||
|
||||
|
||||
class ABTestManager:
|
||||
"""Manage A/B tests."""
|
||||
|
||||
def __init__(self, test_dir: Path = None):
|
||||
self.test_dir = test_dir or Path.home() / ".hermes" / "ab_tests"
|
||||
self.test_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
def create_test(self, task: TestTask) -> ABTestResult:
|
||||
"""Create a new A/B test."""
|
||||
test_id = f"test_{datetime.now().strftime('%Y%m%d_%H%M%S')}_{task.task_id}"
|
||||
result = ABTestResult(
|
||||
test_id=test_id,
|
||||
task=task
|
||||
)
|
||||
self.save_test(result)
|
||||
return result
|
||||
|
||||
def save_test(self, test: ABTestResult):
|
||||
"""Save test results."""
|
||||
path = self.test_dir / f"{test.test_id}.json"
|
||||
with open(path, 'w') as f:
|
||||
json.dump(test.to_dict(), f, indent=2)
|
||||
|
||||
def load_test(self, test_id: str) -> Optional[ABTestResult]:
|
||||
"""Load test results."""
|
||||
path = self.test_dir / f"{test_id}.json"
|
||||
if not path.exists():
|
||||
return None
|
||||
|
||||
try:
|
||||
with open(path, 'r') as f:
|
||||
data = json.load(f)
|
||||
|
||||
task = TestTask(**data["task"])
|
||||
test = ABTestResult(
|
||||
test_id=data["test_id"],
|
||||
task=task,
|
||||
created_at=data.get("created_at", "")
|
||||
)
|
||||
|
||||
for r in data.get("cold_results", []):
|
||||
r["session_type"] = SessionType(r["session_type"])
|
||||
test.cold_results.append(SessionResult(**r))
|
||||
|
||||
for r in data.get("warm_results", []):
|
||||
r["session_type"] = SessionType(r["session_type"])
|
||||
test.warm_results.append(SessionResult(**r))
|
||||
|
||||
return test
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to load test: {e}")
|
||||
return None
|
||||
|
||||
def list_tests(self) -> List[Dict[str, Any]]:
|
||||
"""List all tests."""
|
||||
tests = []
|
||||
for path in self.test_dir.glob("*.json"):
|
||||
try:
|
||||
with open(path, 'r') as f:
|
||||
data = json.load(f)
|
||||
tests.append({
|
||||
"test_id": data.get("test_id"),
|
||||
"task_id": data.get("task", {}).get("task_id"),
|
||||
"description": data.get("task", {}).get("description", ""),
|
||||
"cold_count": len(data.get("cold_results", [])),
|
||||
"warm_count": len(data.get("warm_results", [])),
|
||||
"created_at": data.get("created_at")
|
||||
})
|
||||
except:
|
||||
pass
|
||||
return tests
|
||||
|
||||
def delete_test(self, test_id: str) -> bool:
|
||||
"""Delete a test."""
|
||||
path = self.test_dir / f"{test_id}.json"
|
||||
if path.exists():
|
||||
path.unlink()
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
class ABTestRunner:
|
||||
"""Run A/B tests."""
|
||||
|
||||
def __init__(self, manager: ABTestManager = None):
|
||||
self.manager = manager or ABTestManager()
|
||||
|
||||
def run_comparison(
|
||||
self,
|
||||
task: TestTask,
|
||||
cold_messages: List[Dict],
|
||||
warm_messages: List[Dict],
|
||||
session_db=None
|
||||
) -> Tuple[SessionResult, SessionResult]:
|
||||
"""
|
||||
Run a comparison between cold and warm sessions.
|
||||
|
||||
Returns:
|
||||
Tuple of (cold_result, warm_result)
|
||||
"""
|
||||
# This is a framework - actual execution would depend on
|
||||
# integration with the agent system
|
||||
|
||||
cold_result = SessionResult(
|
||||
session_id=f"cold_{task.task_id}_{int(time.time())}",
|
||||
session_type=SessionType.COLD,
|
||||
task_id=task.task_id,
|
||||
start_time=datetime.now().isoformat()
|
||||
)
|
||||
|
||||
warm_result = SessionResult(
|
||||
session_id=f"warm_{task.task_id}_{int(time.time())}",
|
||||
session_type=SessionType.WARM,
|
||||
task_id=task.task_id,
|
||||
start_time=datetime.now().isoformat()
|
||||
)
|
||||
|
||||
# In a real implementation, this would:
|
||||
# 1. Start a cold session with cold_messages
|
||||
# 2. Execute the task and collect metrics
|
||||
# 3. Start a warm session with warm_messages
|
||||
# 4. Execute the same task and collect metrics
|
||||
# 5. Return both results
|
||||
|
||||
return cold_result, warm_result
|
||||
|
||||
def analyze_results(self, test_id: str) -> Dict[str, Any]:
|
||||
"""Analyze test results."""
|
||||
test = self.manager.load_test(test_id)
|
||||
if not test:
|
||||
return {"error": "Test not found"}
|
||||
|
||||
summary = test.get_summary()
|
||||
|
||||
# Add statistical significance check
|
||||
if (summary["cold"].get("count", 0) >= 3 and
|
||||
summary["warm"].get("count", 0) >= 3):
|
||||
|
||||
# Simple t-test approximation
|
||||
cold_errors = [r.error_rate for r in test.cold_results]
|
||||
warm_errors = [r.error_rate for r in test.warm_results]
|
||||
|
||||
if len(cold_errors) >= 2 and len(warm_errors) >= 2:
|
||||
cold_std = statistics.stdev(cold_errors) if len(cold_errors) > 1 else 0
|
||||
warm_std = statistics.stdev(warm_errors) if len(warm_errors) > 1 else 0
|
||||
|
||||
summary["statistical_notes"] = {
|
||||
"cold_std_dev": cold_std,
|
||||
"warm_std_dev": warm_std,
|
||||
"significance": "low" if max(cold_std, warm_std) > 0.2 else "medium"
|
||||
}
|
||||
|
||||
return summary
|
||||
|
||||
|
||||
# CLI Interface
|
||||
def ab_test_cli(args: List[str]) -> int:
|
||||
"""CLI interface for A/B testing."""
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(description="Warm session A/B testing")
|
||||
subparsers = parser.add_subparsers(dest="command")
|
||||
|
||||
# Create test
|
||||
create_parser = subparsers.add_parser("create", help="Create a new test")
|
||||
create_parser.add_argument("--task-id", required=True, help="Task ID")
|
||||
create_parser.add_argument("--description", required=True, help="Task description")
|
||||
create_parser.add_argument("--prompt", required=True, help="Test prompt")
|
||||
create_parser.add_argument("--category", default="general", help="Task category")
|
||||
create_parser.add_argument("--difficulty", default="medium", choices=["easy", "medium", "hard"])
|
||||
|
||||
# List tests
|
||||
subparsers.add_parser("list", help="List all tests")
|
||||
|
||||
# Show test results
|
||||
show_parser = subparsers.add_parser("show", help="Show test results")
|
||||
show_parser.add_argument("test_id", help="Test ID")
|
||||
|
||||
# Analyze test
|
||||
analyze_parser = subparsers.add_parser("analyze", help="Analyze test results")
|
||||
analyze_parser.add_argument("test_id", help="Test ID")
|
||||
|
||||
# Delete test
|
||||
delete_parser = subparsers.add_parser("delete", help="Delete a test")
|
||||
delete_parser.add_argument("test_id", help="Test ID")
|
||||
|
||||
# Add result
|
||||
add_parser = subparsers.add_parser("add-result", help="Add a test result")
|
||||
add_parser.add_argument("test_id", help="Test ID")
|
||||
add_parser.add_argument("--session-type", required=True, choices=["cold", "warm"])
|
||||
add_parser.add_argument("--session-id", required=True, help="Session ID")
|
||||
add_parser.add_argument("--tool-calls", type=int, default=0)
|
||||
add_parser.add_argument("--successful-calls", type=int, default=0)
|
||||
add_parser.add_argument("--completion-time", type=float, default=0.0)
|
||||
add_parser.add_argument("--success", action="store_true")
|
||||
add_parser.add_argument("--notes", default="")
|
||||
|
||||
parsed = parser.parse_args(args)
|
||||
|
||||
if not parsed.command:
|
||||
parser.print_help()
|
||||
return 1
|
||||
|
||||
manager = ABTestManager()
|
||||
runner = ABTestRunner(manager)
|
||||
|
||||
if parsed.command == "create":
|
||||
task = TestTask(
|
||||
task_id=parsed.task_id,
|
||||
description=parsed.description,
|
||||
prompt=parsed.prompt,
|
||||
category=parsed.category,
|
||||
difficulty=parsed.difficulty
|
||||
)
|
||||
|
||||
test = manager.create_test(task)
|
||||
print(f"Created test: {test.test_id}")
|
||||
print(f"Task: {task.description}")
|
||||
return 0
|
||||
|
||||
elif parsed.command == "list":
|
||||
tests = manager.list_tests()
|
||||
|
||||
if not tests:
|
||||
print("No tests found.")
|
||||
return 0
|
||||
|
||||
print("\n=== A/B Tests ===\n")
|
||||
for t in tests:
|
||||
print(f"ID: {t['test_id']}")
|
||||
print(f" Task: {t['description']}")
|
||||
print(f" Cold tests: {t['cold_count']}, Warm tests: {t['warm_count']}")
|
||||
print(f" Created: {t['created_at']}")
|
||||
print()
|
||||
|
||||
return 0
|
||||
|
||||
elif parsed.command == "show":
|
||||
test = manager.load_test(parsed.test_id)
|
||||
|
||||
if not test:
|
||||
print(f"Test {parsed.test_id} not found")
|
||||
return 1
|
||||
|
||||
print(f"\n=== Test: {test.test_id} ===\n")
|
||||
print(f"Task: {test.task.description}")
|
||||
print(f"Prompt: {test.task.prompt}")
|
||||
print(f"Category: {test.task.category}, Difficulty: {test.task.difficulty}")
|
||||
|
||||
print(f"\nCold sessions: {len(test.cold_results)}")
|
||||
for r in test.cold_results:
|
||||
print(f" {r.session_id}: {r.success_rate:.0%} success, {r.error_rate:.0%} errors")
|
||||
|
||||
print(f"\nWarm sessions: {len(test.warm_results)}")
|
||||
for r in test.warm_results:
|
||||
print(f" {r.session_id}: {r.success_rate:.0%} success, {r.error_rate:.0%} errors")
|
||||
|
||||
return 0
|
||||
|
||||
elif parsed.command == "analyze":
|
||||
analysis = runner.analyze_results(parsed.test_id)
|
||||
|
||||
if "error" in analysis:
|
||||
print(f"Error: {analysis['error']}")
|
||||
return 1
|
||||
|
||||
print(f"\n=== Analysis: {parsed.test_id} ===\n")
|
||||
|
||||
cold = analysis.get("cold", {})
|
||||
warm = analysis.get("warm", {})
|
||||
|
||||
print("Cold Sessions:")
|
||||
print(f" Count: {cold.get('count', 0)}")
|
||||
print(f" Avg error rate: {cold.get('avg_error_rate', 0):.1%}")
|
||||
print(f" Avg success rate: {cold.get('avg_success_rate', 0):.1%}")
|
||||
print(f" Avg completion time: {cold.get('avg_completion_time', 0):.1f}s")
|
||||
|
||||
print("\nWarm Sessions:")
|
||||
print(f" Count: {warm.get('count', 0)}")
|
||||
print(f" Avg error rate: {warm.get('avg_error_rate', 0):.1%}")
|
||||
print(f" Avg success rate: {warm.get('avg_success_rate', 0):.1%}")
|
||||
print(f" Avg completion time: {warm.get('avg_completion_time', 0):.1f}s")
|
||||
|
||||
improvement = analysis.get("improvement", {})
|
||||
if improvement:
|
||||
print("\nImprovement:")
|
||||
if "error_rate" in improvement:
|
||||
print(f" Error rate: {improvement['error_rate']:+.1%}")
|
||||
if "success_rate" in improvement:
|
||||
print(f" Success rate: {improvement['success_rate']:+.1%}")
|
||||
|
||||
print(f"\nRecommendation: {analysis.get('recommendation', 'N/A')}")
|
||||
|
||||
return 0
|
||||
|
||||
elif parsed.command == "delete":
|
||||
if manager.delete_test(parsed.test_id):
|
||||
print(f"Deleted test: {parsed.test_id}")
|
||||
return 0
|
||||
else:
|
||||
print(f"Test {parsed.test_id} not found")
|
||||
return 1
|
||||
|
||||
elif parsed.command == "add-result":
|
||||
test = manager.load_test(parsed.test_id)
|
||||
|
||||
if not test:
|
||||
print(f"Test {parsed.test_id} not found")
|
||||
return 1
|
||||
|
||||
result = SessionResult(
|
||||
session_id=parsed.session_id,
|
||||
session_type=SessionType(parsed.session_type),
|
||||
task_id=test.task.task_id,
|
||||
start_time=datetime.now().isoformat(),
|
||||
end_time=datetime.now().isoformat(),
|
||||
tool_calls=parsed.tool_calls,
|
||||
successful_tool_calls=parsed.successful_calls,
|
||||
completion_time_seconds=parsed.completion_time,
|
||||
success=parsed.success,
|
||||
notes=parsed.notes
|
||||
)
|
||||
|
||||
test.add_result(result)
|
||||
manager.save_test(test)
|
||||
|
||||
print(f"Added {parsed.session_type} result to test {parsed.test_id}")
|
||||
print(f" Session: {parsed.session_id}")
|
||||
print(f" Success rate: {result.success_rate:.0%}")
|
||||
|
||||
return 0
|
||||
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
sys.exit(ab_test_cli(sys.argv[1:]))
|
||||
Reference in New Issue
Block a user