"""Hermes gateway integration — runs, approvals, clarifications, models. This module is the ONLY place that touches Hermes internals, so every Hermes API dependency is documented and defensive (getattr + try/except) to survive normal Hermes upgrades. All Hermes imports are deferred (inside functions) so the module can be imported by unit tests without a Hermes install. """ from __future__ import annotations import asyncio import datetime as dt import logging import threading import time import uuid from pathlib import Path from typing import Any, Dict, List, Optional, Tuple from . import protocol as proto from .conversations import ConversationRouter logger = logging.getLogger(__name__) # Pending interactive requests (approval/clarify) keyed by opaque ID → context. # Single-user app, but a dict keeps the protocol multi-client friendly. _PENDING_APPROVALS: Dict[str, Dict[str, Any]] = {} _PENDING_CLARIFIES: Dict[str, Dict[str, Any]] = {} _RUN_LOCK = threading.Lock() _ACTIVE_RUNS: Dict[str, Dict[str, Any]] = {} # conversation_id → run info _TOOL_EVENTS: Dict[str, Dict[str, Dict[str, Any]]] = {} # The adapter and server are process services, not request-local values. # ContextVars lose their values when Hermes calls plugin hooks from agent # worker threads, which made approvals/tool events disappear. Access is # guarded because hook callbacks can arrive from multiple workers. _SERVICE_LOCK = threading.RLock() _ADAPTER: Any = None _SERVER: Any = None def _runner() -> Any: """The GatewayRunner back-reference injected into the adapter.""" adapter = _current_adapter() return getattr(adapter, "gateway_runner", None) if adapter else None def _current_adapter() -> Any: with _SERVICE_LOCK: return _ADAPTER def set_adapter(adapter: Any) -> None: global _ADAPTER with _SERVICE_LOCK: _ADAPTER = adapter def clear_services(adapter: Any = None) -> None: """Release process-wide references when the owning adapter disconnects.""" global _ADAPTER, _SERVER with _SERVICE_LOCK: if adapter is None or _ADAPTER is adapter: _ADAPTER = None _SERVER = None def _session_store() -> Any: runner = _runner() return getattr(runner, "session_store", None) if runner else None def _session_db() -> Any: runner = _runner() db = getattr(runner, "_session_db", None) if runner else None return getattr(db, "_db", db) if db else None def _source_for(conversation_id: str, user_name: str = "Chris"): """Build the SessionSource for a Pheby conversation (deferred import).""" adapter = _current_adapter() if adapter is not None: return adapter.build_source( chat_id=conversation_id, chat_name=conversation_id, chat_type="dm", user_id="pheby-client", user_name=user_name, ) # Fallback (tests / standalone): construct directly. from gateway.config import Platform from gateway.session import SessionSource return SessionSource( platform=Platform("pheby"), chat_id=str(conversation_id), chat_name=str(conversation_id), chat_type="dm", user_id="pheby-client", user_name=user_name, ) def _session_key_for(conversation_id: str) -> str: """Compute the gateway session key for a conversation. Prefers the SessionStore's own key builder (authoritative); falls back to the documented deterministic shape used by ``build_session_key`` for DM sources (``agent:main::dm:``). """ store = _session_store() if store is not None: try: source = _source_for(conversation_id) return store._generate_session_key(source) except Exception: logger.debug("[pheby] session key via store failed", exc_info=True) return ConversationRouter.session_key_for(conversation_id) def _conversation_from_session_key(session_key: str) -> str: """Extract the opaque chat ID from a Pheby DM session key.""" marker = ":pheby:dm:" if marker in str(session_key): return str(session_key).split(marker, 1)[1] return str(session_key).rsplit(":", 1)[-1] # ═══════════════════════════════════════════════════════════════════════════ # Conversations # ═══════════════════════════════════════════════════════════════════════════ async def create_conversation(server: Any, name: Optional[str]) -> str: """Create the Pheby ID and an empty Hermes session routing entry.""" cid = await server.router.new_conversation(name) store = _session_store() if store is not None: try: await asyncio.to_thread( store.get_or_create_session, _source_for(cid)) if name: await rename_conversation(cid, name) except Exception: # Keep the router entry so the empty conversation remains usable; # the first message can create its Hermes session normally. logger.warning("[pheby] empty Hermes session creation failed", exc_info=True) return cid async def list_conversations(server: Any = None) -> List[Dict[str, Any]]: """Enumerate conversations known to the router + Hermes session store.""" server = server or _current_server() router = server.router if server else None out: List[Dict[str, Any]] = [] seen: set = set() router_names: Dict[str, str] = {} if router is not None: for cid in await router.known_ids(): router_names[cid] = await router.get_name(cid) or cid # 1. Sessions Hermes already tracks for the pheby platform. store = _session_store() if store is not None: try: entries = await asyncio.to_thread(store.list_sessions) for entry in entries: origin = getattr(entry, "origin", None) platform = getattr(getattr(origin, "platform", None), "value", "") if platform != "pheby": continue cid = str(getattr(origin, "chat_id", "") or "") if not cid or cid in seen: continue seen.add(cid) out.append({ "conversation_id": cid, # An explicit Pheby rename wins over Hermes's initial # source-derived display name. "name": (router_names.get(cid) or getattr(entry, "display_name", None) or cid), "session_id": getattr(entry, "session_id", None), "last_active": _iso(getattr(entry, "updated_at", None)), "source": "hermes", }) except Exception: logger.debug("[pheby] session store listing failed", exc_info=True) # 2. Router-known conversations (incl. freshly created, no messages yet). if router is not None: for cid in await router.known_ids(): if cid in seen: continue seen.add(cid) out.append({ "conversation_id": cid, "name": await router.get_name(cid) or cid, "session_id": None, "last_active": None, "source": "pheby", }) out.sort(key=lambda c: (c.get("last_active") is None, c.get("last_active") or ""), reverse=False) out.sort(key=lambda c: c.get("last_active") or "", reverse=True) return out def _iso(value: Any) -> Optional[str]: try: if isinstance(value, (int, float)): return dt.datetime.fromtimestamp(value, tz=dt.timezone.utc).isoformat() if isinstance(value, dt.datetime): # Session-store timestamps are naive LOCAL datetimes (see # gateway.session_lifecycle._now). The client parses with # Instant.parse(), which requires an offset — a bare # "2026-09-10T05:41:48" fails and last_active is lost, so the # app can no longer sort by recency. Attach the local offset # and normalize to UTC. if value.tzinfo is None: value = value.astimezone() return value.astimezone(dt.timezone.utc).isoformat() return value.isoformat() if value else None except (AttributeError, OSError, OverflowError, TypeError, ValueError): return None def _display_user_text(text: str) -> str: """Hide agent-only inline file context from the user-facing transcript.""" start = "[Pheby user message]\n" end = "\n[/Pheby user message]" if start not in text: return text body = text.split(start, 1)[1] return body.split(end, 1)[0] if end in body else text async def conversation_history(conversation_id: str, limit: int ) -> Tuple[List[Dict[str, Any]], bool]: """Load transcript rows for a conversation from Hermes state.db. Returns ``(messages, found)``. ``found`` is False when neither the session store nor the session DB knows the conversation. """ messages: List[Dict[str, Any]] = [] found = False store = _session_store() session_id: Optional[str] = None if store is not None: try: entry = await asyncio.to_thread(store.peek_session_id, _session_key_for(conversation_id)) if entry: session_id = str(entry) found = True except Exception: logger.debug("[pheby] peek_session_id failed", exc_info=True) db = _session_db() if db is not None and session_id: try: rows = await asyncio.to_thread( db.get_messages_as_conversation, session_id, include_row_ids=True) for row in rows[-limit:]: role = row.get("role") if role not in ("user", "assistant"): continue content = row.get("content") text = content if isinstance(content, str) else str(content or "") if role == "user": text = _display_user_text(text) # Tool-call rows can surface as assistant rows with empty # content; skip empties so the client transcript stays clean. if not text.strip() and role == "assistant": continue messages.append({ "message_id": f"m{row.get('_row_id')}" if isinstance(row.get("_row_id"), (int, str)) else None, "role": role, "text": text, "ts": _iso(row.get("timestamp")), }) found = True except Exception: logger.debug("[pheby] transcript load failed", exc_info=True) # A router-known conversation with no messages yet is still "found" so a # fresh client can open it as an empty chat. if not found: server = _current_server() if server is not None: name = await server.router.get_name(conversation_id) if name is not None: found = True return messages, found async def rename_conversation(conversation_id: str, name: str) -> bool: """Rename the Pheby index and the live Hermes routing entry.""" server = _current_server() router = server.router if server else None if router is None: return False if not await router.rename(conversation_id, name): return False store = _session_store() if store is not None: session_key = _session_key_for(conversation_id) def _rename_route() -> None: with store._lock: store._ensure_loaded_locked() entry = store._entries.get(session_key) if entry is not None: entry.display_name = name store._save() try: await asyncio.to_thread(_rename_route) except Exception: logger.debug("[pheby] Hermes display-name update failed", exc_info=True) return True async def delete_conversation(conversation_id: str) -> bool: """Delete a transcript and remove its live routing entry. Hermes currently has no public per-key removal method. We therefore use the same lock/save discipline as SessionStore's own pruning code. Calling ``reset_session`` here would create a replacement entry and make the deleted conversation immediately reappear. """ server = _current_server() router = server.router if server else None if router is None or active_run(conversation_id) is not None: return False store = _session_store() db = _session_db() session_key = _session_key_for(conversation_id) session_id = None if store is not None: try: session_id = await asyncio.to_thread( store.peek_session_id, session_key) except Exception: session_id = None router_known = await router.get_name(conversation_id) is not None if not router_known and not session_id: return False if session_id and db is not None: try: deleted = await asyncio.to_thread(db.delete_session, session_id) if deleted is False: return False except Exception: logger.error("[pheby] session db delete failed", exc_info=True) return False if store is not None: def _remove_route() -> None: with store._lock: store._ensure_loaded_locked() if store._entries.pop(session_key, None) is not None: store._save() try: await asyncio.to_thread(_remove_route) except Exception: logger.error("[pheby] routing removal failed", exc_info=True) return False await router.forget(conversation_id) with _RUN_LOCK: _ACTIVE_RUNS.pop(conversation_id, None) _TOOL_EVENTS.pop(conversation_id, None) return True # ═══════════════════════════════════════════════════════════════════════════ # Chat runs # ═══════════════════════════════════════════════════════════════════════════ async def send_chat(server: Any, conversation_id: str, text: str, client: Any, request_id: Optional[str], attachment_ids: Optional[List[str]] = None) -> None: """Deliver a user message into the Hermes gateway for this conversation. The gateway's full pipeline (auth, sessions, tools, approvals, clarify, deliverables, streaming) runs on the adapter's message handler. Pheby adds nothing to the agent loop. *attachment_ids* (optional) reference inbound uploads already registered in ``server.store``; they are anchored to the user message, handed to Hermes as ``media_urls``/``media_types`` (tool-accessible local files), and small text files are additionally inlined into the message text. """ adapter = _current_adapter() if adapter is None or not hasattr(adapter, "handle_message"): await client.send_json(proto.error_event( proto.ERR_INTERNAL, "Gateway not connected yet", request_id)) return # Register the conversation so it survives restarts. await server.router.ensure_conversation(conversation_id) run_id = uuid.uuid4().hex[:16] source = _source_for(conversation_id) message_id = uuid.uuid4().hex[:12] # ── inbound attachments ──────────────────────────────────────────────── media_urls: List[str] = [] media_types: List[str] = [] media_text_inlined: List[bool] = [] inline_blocks: List[str] = [] attachment_descs: List[Dict[str, Any]] = [] for aid in attachment_ids or []: desc = server.store.describe(aid) if desc is None or desc.get("conversation_id") != conversation_id or \ desc.get("direction") != "inbound" or desc.get("message_id"): await client.send_json(proto.error_event( proto.ERR_NOT_FOUND, "Unknown or already sent attachment for this conversation", request_id)) return attachment_descs.append(desc) for desc in attachment_descs: aid = desc["attachment_id"] blob = server.store.get_blob_path(aid) if blob is None: await client.send_json(proto.error_event( proto.ERR_NOT_FOUND, "Attachment unavailable", request_id)) return media_urls.append(str(blob)) mime = str(desc.get("mime_type", "application/octet-stream")) media_types.append(mime) # Small text files become part of the message text so the model # reads them directly without a tool round-trip. fname = str(desc.get("filename", "file")) inlined = False if mime.startswith("text/") and int(desc.get("size", 0)) <= \ proto.INLINE_TEXT_BYTES: try: content = blob.read_text(encoding="utf-8", errors="replace") inline_blocks.append( f"Attached file: {fname}\n```{fname.rsplit('.', 1)[-1]}\n" f"{content}\n```") inlined = True except OSError: pass media_text_inlined.append(inlined) effective_text = text if attachment_descs: effective_text = f"[Pheby user message]\n{text}\n[/Pheby user message]" if inline_blocks: effective_text += "\n\n" + "\n\n".join(inline_blocks) from gateway.platforms.base import MessageEvent, MessageType event = MessageEvent( text=effective_text, message_type=(MessageType.PHOTO if media_types and all(t.startswith("image/") for t in media_types) else (MessageType.DOCUMENT if media_types else MessageType.TEXT)), source=source, message_id=message_id, metadata={"pheby_run_id": run_id}, media_urls=media_urls, media_types=media_types, media_text_inlined=media_text_inlined, ) with _RUN_LOCK: already_active = conversation_id in _ACTIVE_RUNS if not already_active: _ACTIVE_RUNS[conversation_id] = { "run_id": run_id, "started": asyncio.get_running_loop().time(), } _TOOL_EVENTS[conversation_id] = {} if already_active: await client.send_json(proto.error_event( proto.ERR_RUN_ACTIVE, "A run is already active for this conversation", request_id)) return await client.send_json({ "type": proto.S_RUN_ACCEPTED, "conversation_id": conversation_id, "run_id": run_id, **({"request_id": request_id} if request_id else {}), }) draft_message_id = f"draft-{run_id}" await server.broadcast({ "type": proto.S_MESSAGE_START, "conversation_id": conversation_id, "run_id": run_id, "message_id": draft_message_id, }) # Track the draft in the adapter so send()/edit_message() associate the # final text with the announced draft message id. if getattr(adapter, "_drafts", None) is not None: adapter._drafts.setdefault(conversation_id, { "message_id": draft_message_id, "text": ""}) # The base adapter's handle_message() spawns background tasks and # returns quickly; the eventual reply arrives through adapter.send(). try: await adapter.handle_message(event) for desc in attachment_descs: aid = desc["attachment_id"] server.store.anchor_message(aid, message_id) await server.broadcast({ "type": proto.S_ATTACHMENT_ADDED, "conversation_id": conversation_id, "attachment": server.store.describe(aid), }) except Exception: with _RUN_LOCK: current = _ACTIVE_RUNS.get(conversation_id) if current and current.get("run_id") == run_id: _ACTIVE_RUNS.pop(conversation_id, None) if getattr(adapter, "_drafts", None) is not None: adapter._drafts.pop(conversation_id, None) await server.broadcast({ "type": proto.S_RUN_FINISHED, "conversation_id": conversation_id, "run_id": run_id, "status": "failed", "error": "Gateway rejected the message", }) raise def active_run(conversation_id: str) -> Optional[Dict[str, Any]]: with _RUN_LOCK: run = _ACTIVE_RUNS.get(conversation_id) return dict(run) if run else None def active_run_id(conversation_id: str) -> Optional[str]: run = active_run(conversation_id) return str(run["run_id"]) if run else None def record_tool_event(conversation_id: str, event: Dict[str, Any]) -> None: tool_id = str(event.get("tool_call_id") or "") if not tool_id: return with _RUN_LOCK: bucket = _TOOL_EVENTS.setdefault(conversation_id, {}) bucket[tool_id] = dict(event) if len(bucket) > 200: bucket.pop(next(iter(bucket))) def runtime_snapshot(conversation_id: str) -> Dict[str, Any]: """Recoverable transient state included by ``conversation.open``.""" session_key = _session_key_for(conversation_id) with _RUN_LOCK: run = _ACTIVE_RUNS.get(conversation_id) tools = list(_TOOL_EVENTS.get(conversation_id, {}).values()) approvals = [dict(p["event"]) for p in _PENDING_APPROVALS.values() if p.get("session_key") == session_key] clarifications = [dict(p["event"]) for p in _PENDING_CLARIFIES.values() if p.get("session_key") == session_key] return { "run": dict(run) if run else None, "tools": tools, "approvals": approvals, "clarifications": clarifications, } def note_run_finished(conversation_id: str, status: str = "completed", error: Optional[str] = None) -> None: """Called by the adapter when a turn completes/fails.""" with _RUN_LOCK: run = _ACTIVE_RUNS.pop(conversation_id, None) if run is None: return run_id = run["run_id"] server = _current_server() if server is None: return payload = { "type": proto.S_RUN_FINISHED, "conversation_id": conversation_id, "status": status, } if run_id: payload["run_id"] = run_id if error: payload["error"] = proto.safe_str(error, 300) try: adapter = _current_adapter() if adapter is not None: adapter.schedule_broadcast(payload) except RuntimeError: pass async def cancel_run(conversation_id: str, run_id: Optional[str]) -> bool: """Cancel an active run via Hermes's supported interrupt path.""" adapter = _current_adapter() run = active_run(conversation_id) if run is None: return False if run_id and run["run_id"] != run_id: return False # stale run id — nothing to cancel session_key = _session_key_for(conversation_id) runner = _runner() interrupted = False if runner is not None: # Preferred: gateway's own /stop dispatch (cancels task + drains). running = getattr(runner, "_running_agents", {}).get(session_key) agent = running if running is not None else None interrupt = getattr(agent, "interrupt", None) if callable(interrupt): try: interrupt("Cancelled by Pheby client") invalidate = getattr( runner, "_invalidate_session_run_generation", None) if callable(invalidate): invalidate(session_key, reason="pheby_cancel") interrupted = True except Exception: logger.debug("[pheby] agent interrupt failed", exc_info=True) if not interrupted and adapter is not None: try: event = getattr(adapter, "_active_sessions", {}).get(session_key) if event is not None: await adapter.interrupt_session_activity( session_key, conversation_id) interrupted = True except Exception: logger.debug("[pheby] adapter interrupt failed", exc_info=True) if interrupted: with _RUN_LOCK: _ACTIVE_RUNS.pop(conversation_id, None) return interrupted # ═══════════════════════════════════════════════════════════════════════════ # Approvals # ═══════════════════════════════════════════════════════════════════════════ async def push_approval(approval_data: Dict[str, Any], session_key: str) -> None: """Adapter callback: a dangerous action needs a human decision.""" approval_id = uuid.uuid4().hex[:12] # gateway.run redacts the command before calling send_exec_approval. command = approval_data.get("command", "") choices: List[str] = ["once", "deny"] if approval_data.get("allow_session", True): choices.insert(1, "session") if approval_data.get("allow_permanent", True): choices.insert(-1, "always") conversation_id = _conversation_from_session_key(session_key) event: Dict[str, Any] = { "type": proto.S_APPROVAL_REQUEST, "approval_id": approval_id, "session_key": session_key, "conversation_id": conversation_id, "run_id": active_run_id(conversation_id), "command": proto.safe_str(command, 2000), "description": proto.safe_str( approval_data.get("description", ""), 1000), "choices": choices, "ts": proto.now_iso(), } _PENDING_APPROVALS[approval_id] = { "session_key": session_key, "created": time.monotonic(), "event": event, } server = _current_server() if server is not None: await server.broadcast(event) async def resolve_approval(approval_id: str, choice: str, reason: Optional[str]) -> bool: """Forward an approval decision to Hermes (tools.approval primitives).""" if choice not in ("once", "session", "always", "deny"): return False pending = _PENDING_APPROVALS.get(approval_id) if pending is None: return False try: from tools.approval import resolve_gateway_approval count = await asyncio.to_thread( resolve_gateway_approval, pending["session_key"], choice, False, reason) ok = count > 0 except Exception: logger.error("[pheby] approval resolve failed", exc_info=True) ok = False if ok: _PENDING_APPROVALS.pop(approval_id, None) return ok def fail_stale_approvals(max_age: float = 3600.0) -> None: """Drop approval IDs whose Hermes-side gate has surely timed out.""" now = time.monotonic() for aid in [a for a, p in _PENDING_APPROVALS.items() if now - p["created"] > max_age]: _PENDING_APPROVALS.pop(aid, None) # ═══════════════════════════════════════════════════════════════════════════ # Clarifications # ═══════════════════════════════════════════════════════════════════════════ async def push_clarify(clarify_id: str, session_key: str, question: str, choices: Optional[List[str]]) -> None: """Adapter callback: the agent needs the user to choose.""" conversation_id = _conversation_from_session_key(session_key) event: Dict[str, Any] = { "type": proto.S_CLARIFY_REQUEST, "clarify_id": clarify_id, "session_key": session_key, "conversation_id": conversation_id, "run_id": active_run_id(conversation_id), "question": proto.safe_str(question, 2000), "choices": [proto.safe_str(c, 300) for c in choices] if choices else None, "allow_free_text": True, # Hermes clarify always permits "Other" "ts": proto.now_iso(), } _PENDING_CLARIFIES[clarify_id] = { "session_key": session_key, "created": time.monotonic(), "event": event, } server = _current_server() if server is not None: await server.broadcast(event) async def resolve_clarify(clarify_id: str, response: str) -> bool: """Forward a clarification answer to Hermes's clarify primitive.""" pending = _PENDING_CLARIFIES.get(clarify_id) if pending is None: return False try: from tools.clarify_gateway import resolve_gateway_clarify ok = await asyncio.to_thread( resolve_gateway_clarify, clarify_id, response) if not ok: # Might be an awaiting-text open-ended clarify: route via the # session text path instead. from tools.clarify_gateway import \ resolve_text_response_for_session ok = await asyncio.to_thread( resolve_text_response_for_session, pending["session_key"], response) except Exception: logger.error("[pheby] clarify resolve failed", exc_info=True) ok = False if ok: _PENDING_CLARIFIES.pop(clarify_id, None) return bool(ok) # ═══════════════════════════════════════════════════════════════════════════ # Conversation-scoped YOLO / approval bypass # ═══════════════════════════════════════════════════════════════════════════ async def yolo_snapshot(conversation_id: str) -> Dict[str, Any]: """Return the effective dangerous-command approval bypass for one chat.""" store = _session_store() session_key = _session_key_for(conversation_id) session_id = None if store is not None: try: session_id = await asyncio.to_thread(store.peek_session_id, session_key) except Exception: logger.debug("[pheby] yolo session lookup failed", exc_info=True) if not session_id: return {"ok": False, "code": proto.ERR_CONVERSATION_NOT_FOUND, "message": "Conversation not found"} from tools.approval import enable_session_yolo, is_session_yolo_enabled enabled = is_session_yolo_enabled(session_key) if not enabled: db = _session_db() if db is not None: try: meta = await asyncio.to_thread(db.get_session, str(session_id)) from hermes_state import SessionDB if SessionDB.session_yolo_enabled(meta): enable_session_yolo(session_key) enabled = True except Exception: logger.debug("[pheby] persisted yolo read failed", exc_info=True) return {"ok": True, "enabled": bool(enabled), "scope": "conversation", "conversation_id": conversation_id} async def set_yolo(conversation_id: str, enabled: bool) -> Dict[str, Any]: """Set YOLO for one chat and persist it with that Hermes session.""" store = _session_store() session_key = _session_key_for(conversation_id) session_id = None if store is not None: try: session_id = await asyncio.to_thread(store.peek_session_id, session_key) except Exception: logger.debug("[pheby] yolo session lookup failed", exc_info=True) if not session_id: return {"ok": False, "code": proto.ERR_CONVERSATION_NOT_FOUND, "message": "Conversation not found"} from tools.approval import disable_session_yolo, enable_session_yolo (enable_session_yolo if enabled else disable_session_yolo)(session_key) db = _session_db() if db is not None: try: await asyncio.to_thread(db.set_session_yolo, str(session_id), bool(enabled)) except Exception: logger.warning("[pheby] yolo persistence failed", exc_info=True) return {"ok": True, "enabled": bool(enabled), "scope": "conversation", "conversation_id": conversation_id} # ═══════════════════════════════════════════════════════════════════════════ # Models & reasoning # ═══════════════════════════════════════════════════════════════════════════ async def models_snapshot(conversation_id: Optional[str] = None) -> Dict[str, Any]: """Providers + models Hermes currently exposes (credential-aware).""" def _collect() -> Dict[str, Any]: from hermes_cli.config import get_compatible_custom_providers from hermes_cli.model_switch_providers import list_picker_providers cfg = _load_cfg() model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {} current_model = str(model_cfg.get("default", "") or "") current_provider = str(model_cfg.get("provider", "openrouter") or "") custom_providers = get_compatible_custom_providers(cfg) excluded = (cfg.get("model_catalog") or {}).get( "excluded_providers", []) providers = list_picker_providers( current_provider=current_provider, current_base_url=str(model_cfg.get("base_url", "") or ""), current_model=current_model, user_providers=cfg.get("providers") if isinstance(cfg, dict) else None, custom_providers=custom_providers, excluded_providers=excluded if isinstance(excluded, list) else [], ) return {"providers": providers, "current_model": current_model, "current_provider": current_provider} try: data = await asyncio.to_thread(_collect) except Exception: logger.error("[pheby] model listing failed", exc_info=True) data = {"providers": [], "current_model": "", "current_provider": "", "error": "Model catalog unavailable"} data["supported_reasoning_efforts"] = list(proto.REASONING_EFFORTS) if conversation_id: current = await current_model_snapshot(conversation_id) data["current_model"] = current.get("model", data["current_model"]) data["current_provider"] = current.get( "provider", data["current_provider"]) data["scope"] = current.get("scope", "global") data["ts"] = proto.now_iso() return data async def current_model_snapshot( conversation_id: Optional[str] = None) -> Dict[str, Any]: def _collect() -> Dict[str, Any]: cfg = _load_cfg() model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {} return {"model": str(model_cfg.get("default", "") or ""), "provider": str(model_cfg.get("provider", "") or "")} try: data = await asyncio.to_thread(_collect) except Exception: data = {"model": "", "provider": "", "error": "Config unavailable"} data["scope"] = "global" store = _session_store() if conversation_id and store is not None: try: override = await asyncio.to_thread( store.get_model_override, _session_key_for(conversation_id)) if override: data["model"] = override.get("model", data["model"]) data["provider"] = override.get("provider", data["provider"]) data["scope"] = "conversation" except Exception: logger.debug("[pheby] model override read failed", exc_info=True) data["conversation_id"] = conversation_id data["ts"] = proto.now_iso() return data async def set_model(model: str, provider: Optional[str], conversation_id: Optional[str]) -> Dict[str, Any]: """Change the active model via Hermes's session/global override path.""" if not model: return {"ok": False, "code": proto.ERR_BAD_REQUEST, "message": "model is required"} home_token = None secret_token = None try: from agent.secret_scope import ( build_profile_secret_scope, reset_secret_scope, set_secret_scope, ) from hermes_cli.config import get_compatible_custom_providers from hermes_cli.model_switch import switch_model from hermes_constants import ( get_hermes_home, reset_hermes_home_override, set_hermes_home_override, ) adapter = _current_adapter() profile_home = Path( getattr(adapter, "hermes_home", None) or get_hermes_home()) # model.set is a WebSocket control request, not a gateway message turn, # so the runner has not installed this profile's ContextVars for us. # Bind them explicitly before load_config()/switch_model(); to_thread # copies the current context into its worker. home_token = set_hermes_home_override(str(profile_home)) secrets = await asyncio.to_thread( build_profile_secret_scope, profile_home) secret_token = set_secret_scope(secrets) cfg = _load_cfg() model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {} result = await asyncio.to_thread( switch_model, model, str(model_cfg.get("provider", "openrouter") or "openrouter"), str(model_cfg.get("default", "") or ""), str(model_cfg.get("base_url", "") or ""), "", # current_api_key — scoped runtime resolution handles credentials False, # is_global → session-scoped when conversation given provider or "", cfg.get("providers") if isinstance(cfg, dict) else None, get_compatible_custom_providers(cfg), ) except Exception as exc: logger.error("[pheby] switch_model failed", exc_info=True) return {"ok": False, "code": proto.ERR_BAD_REQUEST, "message": proto.safe_str(exc, 200)} finally: if secret_token is not None: reset_secret_scope(secret_token) if home_token is not None: reset_hermes_home_override(home_token) ok = bool(getattr(result, "success", False)) if not ok: error = (getattr(result, "error_message", "") or getattr(result, "error", "")) return {"ok": False, "code": proto.ERR_BAD_REQUEST, "message": proto.safe_str(error, 300)} resolved_model = (getattr(result, "new_model", "") or getattr(result, "model", "") or model) resolved_provider = (getattr(result, "target_provider", "") or getattr(result, "provider", "") or provider or "") override = {"model": resolved_model} if resolved_provider: override["provider"] = resolved_provider store = _session_store() if conversation_id and store is not None: try: session_key = _session_key_for(conversation_id) if await asyncio.to_thread( store.peek_session_id, session_key) is None: return {"ok": False, "code": proto.ERR_CONVERSATION_NOT_FOUND, "message": "Conversation not found"} await asyncio.to_thread(store.set_model_override, session_key, override) # Match Hermes's native /model commit path: the persisted override # survives restarts, while the richer in-memory override gives the # very next turn its resolved endpoint/key/capabilities. Evict any # cached agent so it cannot answer once more with the old model. runner = _runner() runtime_overrides = getattr( runner, "_session_model_overrides", None) if isinstance(runtime_overrides, dict): runtime_overrides[session_key] = { "model": resolved_model, "provider": resolved_provider, "api_key": getattr(result, "api_key", "") or "", "base_url": getattr(result, "base_url", "") or "", "api_mode": getattr(result, "api_mode", "") or "", "request_overrides": dict( getattr(result, "request_overrides", None) or {}), "capabilities": dict( getattr(result, "runtime_capabilities", None) or {}), } evict = getattr(runner, "_evict_cached_agent", None) if callable(evict): evict(session_key) return {"ok": True, "model": resolved_model, "provider": resolved_provider, "scope": "conversation"} except Exception: logger.debug("[pheby] session model override failed", exc_info=True) # Global fallback: persist via Hermes config save (same path /model # --global uses). try: await asyncio.to_thread(_save_global_model, resolved_model, resolved_provider) return {"ok": True, "model": resolved_model, "provider": resolved_provider, "scope": "global"} except Exception as exc: logger.error("[pheby] global model save failed", exc_info=True) return {"ok": False, "code": proto.ERR_INTERNAL, "message": proto.safe_str(exc, 200)} def _save_global_model(model: str, provider: str) -> None: from hermes_cli.config import save_config_value save_config_value("model.default", model) if provider: save_config_value("model.provider", provider) async def reasoning_snapshot( conversation_id: Optional[str] = None) -> Dict[str, Any]: def _collect() -> Dict[str, Any]: from hermes_constants import resolve_reasoning_config cfg = _load_cfg() model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {} resolved = resolve_reasoning_config( cfg, str(model_cfg.get("default", "") or "")) if resolved is None: return {"effort": None, "enabled": None} if resolved.get("enabled") is False: return {"effort": "none", "enabled": False} return {"effort": resolved.get("effort"), "enabled": True} try: data = await asyncio.to_thread(_collect) except Exception: data = {"effort": None, "enabled": None, "error": "Config unavailable"} data["scope"] = "global" runner = _runner() if conversation_id and runner is not None: try: cfg = await asyncio.to_thread( runner._resolve_session_reasoning_config, session_key=_session_key_for(conversation_id), model="") if cfg is not None: data = ({"effort": "none", "enabled": False} if cfg.get("enabled") is False else {"effort": cfg.get("effort"), "enabled": True}) data["scope"] = "conversation" except Exception: logger.debug("[pheby] reasoning override read failed", exc_info=True) data["conversation_id"] = conversation_id data["supported_efforts"] = ["none"] + list(proto.REASONING_EFFORTS) data["ts"] = proto.now_iso() return data async def set_reasoning(effort: str, conversation_id: Optional[str]) -> Dict[str, Any]: """Set reasoning effort (Hermes levels + 'none' to disable).""" if effort not in ("none",) + proto.REASONING_EFFORTS: return {"ok": False, "code": proto.ERR_BAD_REQUEST, "message": f"effort must be one of: none, " f"{', '.join(proto.REASONING_EFFORTS)}"} parsed = {"enabled": False} if effort == "none" else { "enabled": True, "effort": effort} runner = _runner() if runner is not None and conversation_id: try: store = _session_store() if store is None or await asyncio.to_thread( store.peek_session_id, _session_key_for(conversation_id)) is None: return {"ok": False, "code": proto.ERR_CONVERSATION_NOT_FOUND, "message": "Conversation not found"} await asyncio.to_thread( runner._set_session_reasoning_override, _session_key_for(conversation_id), parsed) return {"ok": True, "effort": effort, "scope": "conversation"} except Exception: logger.debug("[pheby] session reasoning override failed", exc_info=True) try: await asyncio.to_thread(_save_global_reasoning, effort) return {"ok": True, "effort": effort, "scope": "global"} except Exception as exc: return {"ok": False, "code": proto.ERR_INTERNAL, "message": proto.safe_str(exc, 200)} def _save_global_reasoning(effort: str) -> None: from hermes_cli.config import save_config_value save_config_value("agent.reasoning_effort", False if effort == "none" else effort) def _load_cfg() -> Dict[str, Any]: from hermes_cli.config import load_config return load_config() or {} # ═══════════════════════════════════════════════════════════════════════════ # Process service references # ═══════════════════════════════════════════════════════════════════════════ def set_server(server: Any) -> None: global _SERVER with _SERVICE_LOCK: _SERVER = server def _current_server() -> Any: with _SERVICE_LOCK: return _SERVER __all__ = [ "set_adapter", "set_server", "clear_services", "create_conversation", "list_conversations", "conversation_history", "rename_conversation", "delete_conversation", "send_chat", "cancel_run", "note_run_finished", "active_run", "active_run_id", "record_tool_event", "runtime_snapshot", "push_approval", "resolve_approval", "fail_stale_approvals", "push_clarify", "resolve_clarify", "models_snapshot", "current_model_snapshot", "set_model", "reasoning_snapshot", "set_reasoning", ]