"""Hermes gateway integration — runs, approvals, clarifications, models. This module is the ONLY place that touches Hermes internals, so every Hermes API dependency is documented and defensive (getattr + try/except) to survive normal Hermes upgrades. All Hermes imports are deferred (inside functions) so the module can be imported by unit tests without a Hermes install. """ from __future__ import annotations import asyncio import contextvars import logging import threading import uuid from typing import Any, Dict, List, Optional, Tuple from . import protocol as proto from .conversations import ConversationRouter logger = logging.getLogger(__name__) # Pending interactive requests (approval/clarify) keyed by opaque ID → context. # Single-user app, but a dict keeps the protocol multi-client friendly. _PENDING_APPROVALS: Dict[str, Dict[str, Any]] = {} _PENDING_CLARIFIES: Dict[str, Dict[str, Any]] = {} _RUN_LOCK = threading.Lock() _ACTIVE_RUNS: Dict[str, Dict[str, Any]] = {} # conversation_id → run info def _runner() -> Any: """The GatewayRunner back-reference injected into the adapter.""" adapter = _current_adapter() return getattr(adapter, "gateway_runner", None) if adapter else None _ADAPTER_CTX: contextvars.ContextVar = contextvars.ContextVar( "pheby_adapter", default=None) def _current_adapter() -> Any: return _ADAPTER_CTX.get() def set_adapter(adapter: Any) -> None: _ADAPTER_CTX.set(adapter) def _session_store() -> Any: runner = _runner() return getattr(runner, "session_store", None) if runner else None def _session_db() -> Any: runner = _runner() db = getattr(runner, "_session_db", None) if runner else None return getattr(db, "_db", db) if db else None def _source_for(conversation_id: str, user_name: str = "Chris"): """Build the SessionSource for a Pheby conversation (deferred import).""" adapter = _current_adapter() if adapter is not None: return adapter.build_source( chat_id=conversation_id, chat_name=conversation_id, chat_type="dm", user_id="pheby-client", user_name=user_name, ) # Fallback (tests / standalone): construct directly. from gateway.config import Platform from gateway.session import SessionSource return SessionSource( platform=Platform("pheby"), chat_id=str(conversation_id), chat_name=str(conversation_id), chat_type="dm", user_id="pheby-client", user_name=user_name, ) def _session_key_for(conversation_id: str) -> str: """Compute the gateway session key for a conversation. Prefers the SessionStore's own key builder (authoritative); falls back to the documented deterministic shape used by ``build_session_key`` for DM sources (``agent:main::dm:``). """ store = _session_store() if store is not None: try: source = _source_for(conversation_id) return store._generate_session_key(source) except Exception: logger.debug("[pheby] session key via store failed", exc_info=True) return ConversationRouter.session_key_for(conversation_id) # ═══════════════════════════════════════════════════════════════════════════ # Conversations # ═══════════════════════════════════════════════════════════════════════════ async def list_conversations(server: Any = None) -> List[Dict[str, Any]]: """Enumerate conversations known to the router + Hermes session store.""" server = server or _current_server() router = server.router if server else None out: List[Dict[str, Any]] = [] seen: set = set() # 1. Sessions Hermes already tracks for the pheby platform. store = _session_store() if store is not None: try: entries = await asyncio.to_thread(store.list_sessions) for entry in entries: origin = getattr(entry, "origin", None) platform = getattr(getattr(origin, "platform", None), "value", "") if platform != "pheby": continue cid = str(getattr(origin, "chat_id", "") or "") if not cid or cid in seen: continue seen.add(cid) out.append({ "conversation_id": cid, "name": (getattr(entry, "display_name", None) or _router_name(router, cid) or cid), "session_id": getattr(entry, "session_id", None), "last_active": _iso(getattr(entry, "updated_at", None)), "source": "hermes", }) except Exception: logger.debug("[pheby] session store listing failed", exc_info=True) # 2. Router-known conversations (incl. freshly created, no messages yet). if router is not None: for cid in await router.known_ids(): if cid in seen: continue seen.add(cid) out.append({ "conversation_id": cid, "name": await router.get_name(cid) or cid, "session_id": None, "last_active": None, "source": "pheby", }) out.sort(key=lambda c: (c.get("last_active") is None, c.get("last_active") or ""), reverse=False) out.sort(key=lambda c: c.get("last_active") or "", reverse=True) return out def _router_name(router: Any, cid: str) -> Optional[str]: if router is None: return None return router._names.get(cid) def _iso(value: Any) -> Optional[str]: try: return value.isoformat() if value else None except AttributeError: return None async def conversation_history(conversation_id: str, limit: int ) -> Tuple[List[Dict[str, Any]], bool]: """Load transcript rows for a conversation from Hermes state.db. Returns ``(messages, found)``. ``found`` is False when neither the session store nor the session DB knows the conversation. """ messages: List[Dict[str, Any]] = [] found = False store = _session_store() session_id: Optional[str] = None if store is not None: try: source = _source_for(conversation_id) entry = await asyncio.to_thread(store.peek_session_id, _session_key_for(conversation_id)) if entry: session_id = str(entry) found = True except Exception: logger.debug("[pheby] peek_session_id failed", exc_info=True) db = _session_db() if db is not None and session_id: try: rows = await asyncio.to_thread( db.get_messages_as_conversation, session_id) for row in rows[-limit:]: role = row.get("role") if role not in ("user", "assistant"): continue content = row.get("content") text = content if isinstance(content, str) else str(content or "") # Tool-call rows can surface as assistant rows with empty # content; skip empties so the client transcript stays clean. if not text.strip() and role == "assistant": continue messages.append({ "message_id": f"m{row.get('id', len(messages))}" if isinstance(row.get("id"), (int, str)) else None, "role": role, "text": text, "ts": row.get("timestamp") if isinstance( row.get("timestamp"), str) else None, }) found = True except Exception: logger.debug("[pheby] transcript load failed", exc_info=True) # A router-known conversation with no messages yet is still "found" so a # fresh client can open it as an empty chat. if not found: server = _current_server() if server is not None: name = await server.router.get_name(conversation_id) if name is not None: found = True return messages, found async def delete_conversation(conversation_id: str) -> bool: """Delete a conversation from the router + Hermes (best effort on DB). Hermes limitation: the SessionStore has no public per-key delete; the authoritative delete is ``SessionDB.delete_session`` on the current session id. The routing entry is also reset so the next message starts a fresh session. Documented approximation — see README limitations. """ server = _current_server() router = server.router if server else None if router is None: return False if not await router.forget(conversation_id): return False store = _session_store() db = _session_db() session_id = None if store is not None: try: session_id = await asyncio.to_thread( store.peek_session_id, _session_key_for(conversation_id)) except Exception: session_id = None if session_id and db is not None: try: await asyncio.to_thread(db.delete_session, session_id) except Exception: logger.debug("[pheby] session db delete failed", exc_info=True) if store is not None: try: await asyncio.to_thread(store.reset_session, _session_key_for(conversation_id), None) except Exception: logger.debug("[pheby] store reset failed", exc_info=True) _ACTIVE_RUNS.pop(conversation_id, None) return True # ═══════════════════════════════════════════════════════════════════════════ # Chat runs # ═══════════════════════════════════════════════════════════════════════════ async def send_chat(server: Any, conversation_id: str, text: str, client: Any, request_id: Optional[str]) -> None: """Deliver a user message into the Hermes gateway for this conversation. The gateway's full pipeline (auth, sessions, tools, approvals, clarify, deliverables, streaming) runs on the adapter's message handler. Pheby adds nothing to the agent loop. """ adapter = _current_adapter() if adapter is None or not hasattr(adapter, "handle_message"): await client.send_json(proto.error_event( proto.ERR_INTERNAL, "Gateway not connected yet", request_id)) return # Register the conversation so it survives restarts. await server.router.ensure_conversation(conversation_id) run_id = uuid.uuid4().hex[:16] source = _source_for(conversation_id) from gateway.platforms.base import MessageEvent, MessageType event = MessageEvent( text=text, message_type=MessageType.TEXT, source=source, message_id=uuid.uuid4().hex[:12], metadata={"pheby_run_id": run_id}, ) _ACTIVE_RUNS[conversation_id] = { "run_id": run_id, "started": asyncio.get_event_loop().time(), } await client.send_json({ "type": proto.S_RUN_ACCEPTED, "conversation_id": conversation_id, "run_id": run_id, **({"request_id": request_id} if request_id else {}), }) draft_message_id = f"draft-{run_id}" await server.broadcast({ "type": proto.S_MESSAGE_START, "conversation_id": conversation_id, "run_id": run_id, "message_id": draft_message_id, }) # Track the draft in the adapter so send()/edit_message() associate the # final text with the announced draft message id. if getattr(adapter, "_drafts", None) is not None: adapter._drafts.setdefault(conversation_id, { "message_id": draft_message_id, "text": ""}) # The base adapter's handle_message() spawns background tasks and # returns quickly; the eventual reply arrives through adapter.send(). await adapter.handle_message(event) def note_run_finished(conversation_id: str, status: str = "completed", error: Optional[str] = None) -> None: """Called by the adapter when a turn completes/fails.""" run = _ACTIVE_RUNS.pop(conversation_id, None) run_id = run["run_id"] if run else None server = _current_server() if server is None: return payload = { "type": proto.S_RUN_FINISHED, "conversation_id": conversation_id, "status": status, } if run_id: payload["run_id"] = run_id if error: payload["error"] = proto.safe_str(error, 300) try: loop = asyncio.get_event_loop() if loop.is_running(): asyncio.ensure_future(server.broadcast(payload)) except RuntimeError: pass async def cancel_run(conversation_id: str, run_id: Optional[str]) -> bool: """Cancel an active run via Hermes's supported interrupt path.""" adapter = _current_adapter() run = _ACTIVE_RUNS.get(conversation_id) if run and run_id and run["run_id"] != run_id: return False # stale run id — nothing to cancel session_key = _session_key_for(conversation_id) runner = _runner() interrupted = False if runner is not None: # Preferred: gateway's own /stop dispatch (cancels task + drains). running = getattr(runner, "_running_agents", {}).get(session_key) agent = running if running is not None else None if agent is not None and agent is not getattr( type(runner), "_AGENT_PENDING_SENTINEL", object()): try: agent.interrupt("Cancelled by Pheby client") invalidate = getattr( runner, "_invalidate_session_run_generation", None) if callable(invalidate): invalidate(session_key, reason="pheby_cancel") interrupted = True except Exception: logger.debug("[pheby] agent interrupt failed", exc_info=True) if not interrupted and adapter is not None: try: await adapter.interrupt_session_activity( session_key, conversation_id) interrupted = True except Exception: logger.debug("[pheby] adapter interrupt failed", exc_info=True) note_run_finished(conversation_id, "cancelled" if interrupted else "idle") return interrupted # ═══════════════════════════════════════════════════════════════════════════ # Approvals # ═══════════════════════════════════════════════════════════════════════════ async def push_approval(approval_data: Dict[str, Any], session_key: str) -> None: """Adapter callback: a dangerous action needs a human decision.""" approval_id = uuid.uuid4().hex[:12] from gateway.run import _redact_approval_command command = _redact_approval_command(approval_data.get("command", "")) choices: List[str] = ["once", "deny"] if approval_data.get("allow_session", True): choices.insert(1, "session") if approval_data.get("allow_permanent", True): choices.insert(-1, "always") event = { "type": proto.S_APPROVAL_REQUEST, "approval_id": approval_id, "session_key": session_key, "command": proto.safe_str(command, 2000), "description": proto.safe_str( approval_data.get("description", ""), 1000), "choices": choices, "ts": proto.now_iso(), } _PENDING_APPROVALS[approval_id] = { "session_key": session_key, "created": asyncio.get_event_loop().time(), } server = _current_server() if server is not None: await server.broadcast(event) async def resolve_approval(approval_id: str, choice: str, reason: Optional[str]) -> bool: """Forward an approval decision to Hermes (tools.approval primitives).""" pending = _PENDING_APPROVALS.pop(approval_id, None) if pending is None: return False if choice not in ("once", "session", "always", "deny"): return False try: from tools.approval import resolve_gateway_approval count = await asyncio.to_thread( resolve_gateway_approval, pending["session_key"], choice, False, reason) ok = count > 0 except Exception: logger.error("[pheby] approval resolve failed", exc_info=True) ok = False server = _current_server() if server is not None: await server.broadcast({ "type": proto.S_APPROVAL_RESOLVED, "approval_id": approval_id, "choice": choice, "accepted": ok, }) return True def fail_stale_approvals(max_age: float = 3600.0) -> None: """Drop approval IDs whose Hermes-side gate has surely timed out.""" now = asyncio.get_event_loop().time() for aid in [a for a, p in _PENDING_APPROVALS.items() if now - p["created"] > max_age]: _PENDING_APPROVALS.pop(aid, None) # ═══════════════════════════════════════════════════════════════════════════ # Clarifications # ═══════════════════════════════════════════════════════════════════════════ async def push_clarify(clarify_id: str, session_key: str, question: str, choices: Optional[List[str]]) -> None: """Adapter callback: the agent needs the user to choose.""" event = { "type": proto.S_CLARIFY_REQUEST, "clarify_id": clarify_id, "session_key": session_key, "question": proto.safe_str(question, 2000), "choices": [proto.safe_str(c, 300) for c in choices] if choices else None, "allow_free_text": True, # Hermes clarify always permits "Other" "ts": proto.now_iso(), } _PENDING_CLARIFIES[clarify_id] = { "session_key": session_key, "created": asyncio.get_event_loop().time(), } server = _current_server() if server is not None: await server.broadcast(event) async def resolve_clarify(clarify_id: str, response: str) -> bool: """Forward a clarification answer to Hermes's clarify primitive.""" pending = _PENDING_CLARIFIES.pop(clarify_id, None) if pending is None: return False try: from tools.clarify_gateway import resolve_gateway_clarify ok = await asyncio.to_thread( resolve_gateway_clarify, clarify_id, response) if not ok: # Might be an awaiting-text open-ended clarify: route via the # session text path instead. from tools.clarify_gateway import \ resolve_text_response_for_session ok = await asyncio.to_thread( resolve_text_response_for_session, pending["session_key"], response) except Exception: logger.error("[pheby] clarify resolve failed", exc_info=True) ok = False server = _current_server() if server is not None: await server.broadcast({ "type": proto.S_CLARIFY_RESOLVED, "clarify_id": clarify_id, "accepted": bool(ok), }) return bool(ok) # ═══════════════════════════════════════════════════════════════════════════ # Models & reasoning # ═══════════════════════════════════════════════════════════════════════════ async def models_snapshot() -> Dict[str, Any]: """Providers + models Hermes currently exposes (credential-aware).""" def _collect() -> Dict[str, Any]: from hermes_cli.model_switch import list_picker_providers cfg = _load_cfg() model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {} current_model = str(model_cfg.get("default", "") or "") current_provider = str(model_cfg.get("provider", "openrouter") or "") providers = list_picker_providers( current_provider=current_provider, current_model=current_model, user_providers=cfg.get("providers") if isinstance(cfg, dict) else None, probe_custom_providers=False, # don't block on offline endpoints ) return {"providers": providers, "current_model": current_model, "current_provider": current_provider} try: data = await asyncio.to_thread(_collect) except Exception: logger.error("[pheby] model listing failed", exc_info=True) data = {"providers": [], "current_model": "", "current_provider": "", "error": "Model catalog unavailable"} data["supported_reasoning_efforts"] = list(proto.REASONING_EFFORTS) data["ts"] = proto.now_iso() return data async def current_model_snapshot() -> Dict[str, Any]: def _collect() -> Dict[str, Any]: cfg = _load_cfg() model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {} return {"model": str(model_cfg.get("default", "") or ""), "provider": str(model_cfg.get("provider", "") or "")} try: data = await asyncio.to_thread(_collect) except Exception: data = {"model": "", "provider": "", "error": "Config unavailable"} data["ts"] = proto.now_iso() return data async def set_model(model: str, provider: Optional[str], conversation_id: Optional[str]) -> Dict[str, Any]: """Change the active model via Hermes's session/global override path.""" if not model: return {"ok": False, "code": proto.ERR_BAD_REQUEST, "message": "model is required"} try: from hermes_cli.model_switch import switch_model cfg = _load_cfg() model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {} result = await asyncio.to_thread( switch_model, model, str(model_cfg.get("provider", "openrouter") or "openrouter"), str(model_cfg.get("default", "") or ""), str(model_cfg.get("base_url", "") or ""), "", # current_api_key — runtime resolution handles credentials False, # is_global → session-scoped when conversation given provider or "", cfg.get("providers") if isinstance(cfg, dict) else None, None, ) except Exception as exc: logger.error("[pheby] switch_model failed", exc_info=True) return {"ok": False, "code": proto.ERR_BAD_REQUEST, "message": proto.safe_str(exc, 200)} ok = bool(getattr(result, "success", False)) if not ok: return {"ok": False, "code": proto.ERR_BAD_REQUEST, "message": proto.safe_str(getattr(result, "error", ""), 300)} resolved_model = getattr(result, "model", model) resolved_provider = getattr(result, "provider", provider or "") override = {"model": resolved_model} if resolved_provider: override["provider"] = resolved_provider store = _session_store() if conversation_id and store is not None: try: await asyncio.to_thread(store.set_model_override, _session_key_for(conversation_id), override) return {"ok": True, "model": resolved_model, "provider": resolved_provider, "scope": "conversation"} except Exception: logger.debug("[pheby] session model override failed", exc_info=True) # Global fallback: persist via Hermes config save (same path /model # --global uses). try: await asyncio.to_thread(_save_global_model, resolved_model, resolved_provider) return {"ok": True, "model": resolved_model, "provider": resolved_provider, "scope": "global"} except Exception as exc: logger.error("[pheby] global model save failed", exc_info=True) return {"ok": False, "code": proto.ERR_INTERNAL, "message": proto.safe_str(exc, 200)} def _save_global_model(model: str, provider: str) -> None: from hermes_cli.config import load_config, save_config_value save_config_value("model.default", model) if provider: save_config_value("model.provider", provider) async def reasoning_snapshot() -> Dict[str, Any]: def _collect() -> Dict[str, Any]: from hermes_constants import resolve_reasoning_config cfg = _load_cfg() model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {} resolved = resolve_reasoning_config( cfg, str(model_cfg.get("default", "") or "")) if resolved is None: return {"effort": None, "enabled": None} if resolved.get("enabled") is False: return {"effort": "none", "enabled": False} return {"effort": resolved.get("effort"), "enabled": True} try: data = await asyncio.to_thread(_collect) except Exception: data = {"effort": None, "enabled": None, "error": "Config unavailable"} data["supported_efforts"] = ["none"] + list(proto.REASONING_EFFORTS) data["ts"] = proto.now_iso() return data async def set_reasoning(effort: str, conversation_id: Optional[str]) -> Dict[str, Any]: """Set reasoning effort (Hermes levels + 'none' to disable).""" if effort not in ("none",) + proto.REASONING_EFFORTS: return {"ok": False, "code": proto.ERR_BAD_REQUEST, "message": f"effort must be one of: none, " f"{', '.join(proto.REASONING_EFFORTS)}"} parsed = {"enabled": False} if effort == "none" else { "enabled": True, "effort": effort} runner = _runner() if runner is not None and conversation_id: try: await asyncio.to_thread( runner._set_session_reasoning_override, _session_key_for(conversation_id), parsed) return {"ok": True, "effort": effort, "scope": "conversation"} except Exception: logger.debug("[pheby] session reasoning override failed", exc_info=True) try: await asyncio.to_thread(_save_global_reasoning, effort) return {"ok": True, "effort": effort, "scope": "global"} except Exception as exc: return {"ok": False, "code": proto.ERR_INTERNAL, "message": proto.safe_str(exc, 200)} def _save_global_reasoning(effort: str) -> None: from hermes_cli.config import save_config_value save_config_value("agent.reasoning_effort", False if effort == "none" else effort) def _load_cfg() -> Dict[str, Any]: from hermes_cli.config import load_config return load_config() or {} # ═══════════════════════════════════════════════════════════════════════════ # Server context # ═══════════════════════════════════════════════════════════════════════════ _SERVER_CTX: contextvars.ContextVar = contextvars.ContextVar( "pheby_server", default=None) def set_server(server: Any) -> None: _SERVER_CTX.set(server) def _current_server() -> Any: return _SERVER_CTX.get() __all__ = [ "set_adapter", "set_server", "list_conversations", "conversation_history", "delete_conversation", "send_chat", "cancel_run", "note_run_finished", "push_approval", "resolve_approval", "fail_stale_approvals", "push_clarify", "resolve_clarify", "models_snapshot", "current_model_snapshot", "set_model", "reasoning_snapshot", "set_reasoning", ]