[loop-cycle-946] refactor: complete airllm removal (#486) (#545)

2026-03-19 20:46:20 -04:00
parent 88e59f7c17
commit 7da434c85b
10 changed files with 17 additions and 553 deletions
--- a/src/timmy/agent.py
+++ b/src/timmy/agent.py
@@ -26,12 +26,12 @@ from timmy.prompts import get_system_prompt
 from timmy.tools import create_full_toolkit

 if TYPE_CHECKING:
-    from timmy.backends import ClaudeBackend, GrokBackend, TimmyAirLLMAgent
+    from timmy.backends import ClaudeBackend, GrokBackend

 logger = logging.getLogger(__name__)

 # Union type for callers that want to hint the return type.
-TimmyAgent = Union[Agent, "TimmyAirLLMAgent", "GrokBackend", "ClaudeBackend"]
+TimmyAgent = Union[Agent, "GrokBackend", "ClaudeBackend"]

 # Models known to be too small for reliable tool calling.
 # These hallucinate tool calls as text, invoke tools randomly,
@@ -172,29 +172,17 @@ def _warmup_model(model_name: str) -> bool:


 def _resolve_backend(requested: str | None) -> str:
-    """Return the backend name to use, resolving 'auto' and explicit overrides.
+    """Return the backend name to use.

-    Priority (highest → lowest):
+    Priority (highest -> lowest):
      1. CLI flag passed directly to create_timmy()
      2. TIMMY_MODEL_BACKEND env var / .env setting
-      3. 'ollama' (safe default — no surprises)
-
-    'auto' triggers Apple Silicon detection: uses AirLLM if both
-    is_apple_silicon() and airllm_available() return True.
+      3. 'ollama' (safe default -- no surprises)
    """
    if requested is not None:
        return requested

-    configured = settings.timmy_model_backend  # "ollama" | "airllm" | "grok" | "claude" | "auto"
-    if configured != "auto":
-        return configured
-
-    # "auto" path — lazy import to keep startup fast and tests clean.
-    from timmy.backends import airllm_available, is_apple_silicon
-
-    if is_apple_silicon() and airllm_available():
-        return "airllm"
-    return "ollama"
+    return settings.timmy_model_backend  # "ollama" | "grok" | "claude"


 def _build_tools_list(use_tools: bool, skip_mcp: bool, model_name: str) -> list:
@@ -284,17 +272,15 @@ def _create_ollama_agent(
 def create_timmy(
    db_file: str = "timmy.db",
    backend: str | None = None,
-    model_size: str | None = None,
    *,
    skip_mcp: bool = False,
    session_id: str = "unknown",
 ) -> TimmyAgent:
-    """Instantiate the agent — Ollama or AirLLM, same public interface.
+    """Instantiate the agent — Ollama, Grok, or Claude.

    Args:
        db_file:    SQLite file for Agno conversation memory (Ollama path only).
-        backend:    "ollama" | "airllm" | "auto" | None (reads config/env).
-        model_size: AirLLM size — "8b" | "70b" | "405b" | None (reads config).
+        backend:    "ollama" | "grok" | "claude" | None (reads config/env).
        skip_mcp:   If True, omit MCP tool servers (Gitea, filesystem).
                    Use for background tasks (thinking, QA) where MCP's
                    stdio cancel-scope lifecycle conflicts with asyncio
@@ -304,7 +290,6 @@ def create_timmy(
    print_response(message, stream).
    """
    resolved = _resolve_backend(backend)
-    size = model_size or "70b"

    if resolved == "claude":
        from timmy.backends import ClaudeBackend
@@ -316,11 +301,6 @@ def create_timmy(

        return GrokBackend()

-    if resolved == "airllm":
-        from timmy.backends import TimmyAirLLMAgent
-
-        return TimmyAirLLMAgent(model_size=size)
-
    # Default: Ollama via Agno.
    model_name, is_fallback = _resolve_model_with_fallback(
        requested_model=None,