1135 lines
47 KiB
Python
1135 lines
47 KiB
Python
"""Hermes gateway integration — runs, approvals, clarifications, models.
|
|
|
|
This module is the ONLY place that touches Hermes internals, so every Hermes
|
|
API dependency is documented and defensive (getattr + try/except) to survive
|
|
normal Hermes upgrades. All Hermes imports are deferred (inside functions)
|
|
so the module can be imported by unit tests without a Hermes install.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import datetime as dt
|
|
import logging
|
|
import threading
|
|
import time
|
|
import uuid
|
|
from pathlib import Path
|
|
from typing import Any, Dict, List, Optional, Tuple
|
|
|
|
from . import protocol as proto
|
|
from .conversations import ConversationRouter
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Pending interactive requests (approval/clarify) keyed by opaque ID → context.
|
|
# Single-user app, but a dict keeps the protocol multi-client friendly.
|
|
_PENDING_APPROVALS: Dict[str, Dict[str, Any]] = {}
|
|
_PENDING_CLARIFIES: Dict[str, Dict[str, Any]] = {}
|
|
|
|
_RUN_LOCK = threading.Lock()
|
|
_ACTIVE_RUNS: Dict[str, Dict[str, Any]] = {} # conversation_id → run info
|
|
_TOOL_EVENTS: Dict[str, Dict[str, Dict[str, Any]]] = {}
|
|
|
|
# The adapter and server are process services, not request-local values.
|
|
# ContextVars lose their values when Hermes calls plugin hooks from agent
|
|
# worker threads, which made approvals/tool events disappear. Access is
|
|
# guarded because hook callbacks can arrive from multiple workers.
|
|
_SERVICE_LOCK = threading.RLock()
|
|
_ADAPTER: Any = None
|
|
_SERVER: Any = None
|
|
|
|
|
|
def _runner() -> Any:
|
|
"""The GatewayRunner back-reference injected into the adapter."""
|
|
adapter = _current_adapter()
|
|
return getattr(adapter, "gateway_runner", None) if adapter else None
|
|
|
|
|
|
def _current_adapter() -> Any:
|
|
with _SERVICE_LOCK:
|
|
return _ADAPTER
|
|
|
|
|
|
def set_adapter(adapter: Any) -> None:
|
|
global _ADAPTER
|
|
with _SERVICE_LOCK:
|
|
_ADAPTER = adapter
|
|
|
|
|
|
def clear_services(adapter: Any = None) -> None:
|
|
"""Release process-wide references when the owning adapter disconnects."""
|
|
global _ADAPTER, _SERVER
|
|
with _SERVICE_LOCK:
|
|
if adapter is None or _ADAPTER is adapter:
|
|
_ADAPTER = None
|
|
_SERVER = None
|
|
|
|
|
|
def _session_store() -> Any:
|
|
runner = _runner()
|
|
return getattr(runner, "session_store", None) if runner else None
|
|
|
|
|
|
def _session_db() -> Any:
|
|
runner = _runner()
|
|
db = getattr(runner, "_session_db", None) if runner else None
|
|
return getattr(db, "_db", db) if db else None
|
|
|
|
|
|
def _source_for(conversation_id: str, user_name: str = "Chris"):
|
|
"""Build the SessionSource for a Pheby conversation (deferred import)."""
|
|
adapter = _current_adapter()
|
|
if adapter is not None:
|
|
return adapter.build_source(
|
|
chat_id=conversation_id,
|
|
chat_name=conversation_id,
|
|
chat_type="dm",
|
|
user_id="pheby-client",
|
|
user_name=user_name,
|
|
)
|
|
# Fallback (tests / standalone): construct directly.
|
|
from gateway.config import Platform
|
|
from gateway.session import SessionSource
|
|
return SessionSource(
|
|
platform=Platform("pheby"),
|
|
chat_id=str(conversation_id),
|
|
chat_name=str(conversation_id),
|
|
chat_type="dm",
|
|
user_id="pheby-client",
|
|
user_name=user_name,
|
|
)
|
|
|
|
|
|
def _session_key_for(conversation_id: str) -> str:
|
|
"""Compute the gateway session key for a conversation.
|
|
|
|
Prefers the SessionStore's own key builder (authoritative); falls back to
|
|
the documented deterministic shape used by ``build_session_key`` for DM
|
|
sources (``agent:main:<platform>:dm:<chat_id>``).
|
|
"""
|
|
store = _session_store()
|
|
if store is not None:
|
|
try:
|
|
source = _source_for(conversation_id)
|
|
return store._generate_session_key(source)
|
|
except Exception:
|
|
logger.debug("[pheby] session key via store failed", exc_info=True)
|
|
return ConversationRouter.session_key_for(conversation_id)
|
|
|
|
|
|
def _conversation_from_session_key(session_key: str) -> str:
|
|
"""Extract the opaque chat ID from a Pheby DM session key."""
|
|
marker = ":pheby:dm:"
|
|
if marker in str(session_key):
|
|
return str(session_key).split(marker, 1)[1]
|
|
return str(session_key).rsplit(":", 1)[-1]
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
# Conversations
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
async def create_conversation(server: Any, name: Optional[str]) -> str:
|
|
"""Create the Pheby ID and an empty Hermes session routing entry."""
|
|
cid = await server.router.new_conversation(name)
|
|
store = _session_store()
|
|
if store is not None:
|
|
try:
|
|
await asyncio.to_thread(
|
|
store.get_or_create_session, _source_for(cid))
|
|
if name:
|
|
await rename_conversation(cid, name)
|
|
except Exception:
|
|
# Keep the router entry so the empty conversation remains usable;
|
|
# the first message can create its Hermes session normally.
|
|
logger.warning("[pheby] empty Hermes session creation failed",
|
|
exc_info=True)
|
|
return cid
|
|
|
|
|
|
async def list_conversations(server: Any = None) -> List[Dict[str, Any]]:
|
|
"""Enumerate conversations known to the router + Hermes session store."""
|
|
server = server or _current_server()
|
|
router = server.router if server else None
|
|
out: List[Dict[str, Any]] = []
|
|
seen: set = set()
|
|
router_names: Dict[str, str] = {}
|
|
if router is not None:
|
|
for cid in await router.known_ids():
|
|
router_names[cid] = await router.get_name(cid) or cid
|
|
|
|
# 1. Sessions Hermes already tracks for the pheby platform.
|
|
store = _session_store()
|
|
if store is not None:
|
|
try:
|
|
entries = await asyncio.to_thread(store.list_sessions)
|
|
for entry in entries:
|
|
origin = getattr(entry, "origin", None)
|
|
platform = getattr(getattr(origin, "platform", None),
|
|
"value", "")
|
|
if platform != "pheby":
|
|
continue
|
|
cid = str(getattr(origin, "chat_id", "") or "")
|
|
if not cid or cid in seen:
|
|
continue
|
|
seen.add(cid)
|
|
out.append({
|
|
"conversation_id": cid,
|
|
# An explicit Pheby rename wins over Hermes's initial
|
|
# source-derived display name.
|
|
"name": (router_names.get(cid)
|
|
or getattr(entry, "display_name", None) or cid),
|
|
"session_id": getattr(entry, "session_id", None),
|
|
"last_active": _iso(getattr(entry, "updated_at", None)),
|
|
"source": "hermes",
|
|
})
|
|
except Exception:
|
|
logger.debug("[pheby] session store listing failed", exc_info=True)
|
|
|
|
# 2. Router-known conversations (incl. freshly created, no messages yet).
|
|
if router is not None:
|
|
for cid in await router.known_ids():
|
|
if cid in seen:
|
|
continue
|
|
seen.add(cid)
|
|
out.append({
|
|
"conversation_id": cid,
|
|
"name": await router.get_name(cid) or cid,
|
|
"session_id": None,
|
|
"last_active": None,
|
|
"source": "pheby",
|
|
})
|
|
|
|
out.sort(key=lambda c: (c.get("last_active") is None,
|
|
c.get("last_active") or ""), reverse=False)
|
|
out.sort(key=lambda c: c.get("last_active") or "", reverse=True)
|
|
return out
|
|
|
|
|
|
def _iso(value: Any) -> Optional[str]:
|
|
try:
|
|
if isinstance(value, (int, float)):
|
|
return dt.datetime.fromtimestamp(value, tz=dt.timezone.utc).isoformat()
|
|
if isinstance(value, dt.datetime):
|
|
# Session-store timestamps are naive LOCAL datetimes (see
|
|
# gateway.session_lifecycle._now). The client parses with
|
|
# Instant.parse(), which requires an offset — a bare
|
|
# "2026-09-10T05:41:48" fails and last_active is lost, so the
|
|
# app can no longer sort by recency. Attach the local offset
|
|
# and normalize to UTC.
|
|
if value.tzinfo is None:
|
|
value = value.astimezone()
|
|
return value.astimezone(dt.timezone.utc).isoformat()
|
|
return value.isoformat() if value else None
|
|
except (AttributeError, OSError, OverflowError, TypeError, ValueError):
|
|
return None
|
|
|
|
|
|
def _display_user_text(text: str) -> str:
|
|
"""Hide agent-only inline file context from the user-facing transcript."""
|
|
start = "[Pheby user message]\n"
|
|
end = "\n[/Pheby user message]"
|
|
if start not in text:
|
|
return text
|
|
body = text.split(start, 1)[1]
|
|
return body.split(end, 1)[0] if end in body else text
|
|
|
|
|
|
async def conversation_history(conversation_id: str, limit: int
|
|
) -> Tuple[List[Dict[str, Any]], bool]:
|
|
"""Load transcript rows for a conversation from Hermes state.db.
|
|
|
|
Returns ``(messages, found)``. ``found`` is False when neither the
|
|
session store nor the session DB knows the conversation.
|
|
"""
|
|
messages: List[Dict[str, Any]] = []
|
|
found = False
|
|
|
|
store = _session_store()
|
|
session_id: Optional[str] = None
|
|
if store is not None:
|
|
try:
|
|
entry = await asyncio.to_thread(store.peek_session_id,
|
|
_session_key_for(conversation_id))
|
|
if entry:
|
|
session_id = str(entry)
|
|
found = True
|
|
except Exception:
|
|
logger.debug("[pheby] peek_session_id failed", exc_info=True)
|
|
|
|
db = _session_db()
|
|
if db is not None and session_id:
|
|
try:
|
|
rows = await asyncio.to_thread(
|
|
db.get_messages_as_conversation, session_id,
|
|
include_row_ids=True)
|
|
for row in rows[-limit:]:
|
|
role = row.get("role")
|
|
if role not in ("user", "assistant"):
|
|
continue
|
|
content = row.get("content")
|
|
text = content if isinstance(content, str) else str(content or "")
|
|
if role == "user":
|
|
text = _display_user_text(text)
|
|
# Tool-call rows can surface as assistant rows with empty
|
|
# content; skip empties so the client transcript stays clean.
|
|
if not text.strip() and role == "assistant":
|
|
continue
|
|
messages.append({
|
|
"message_id": f"m{row.get('_row_id')}"
|
|
if isinstance(row.get("_row_id"), (int, str)) else None,
|
|
"role": role,
|
|
"text": text,
|
|
"ts": _iso(row.get("timestamp")),
|
|
})
|
|
found = True
|
|
except Exception:
|
|
logger.debug("[pheby] transcript load failed", exc_info=True)
|
|
|
|
# A router-known conversation with no messages yet is still "found" so a
|
|
# fresh client can open it as an empty chat.
|
|
if not found:
|
|
server = _current_server()
|
|
if server is not None:
|
|
name = await server.router.get_name(conversation_id)
|
|
if name is not None:
|
|
found = True
|
|
return messages, found
|
|
|
|
|
|
async def rename_conversation(conversation_id: str, name: str) -> bool:
|
|
"""Rename the Pheby index and the live Hermes routing entry."""
|
|
server = _current_server()
|
|
router = server.router if server else None
|
|
if router is None:
|
|
return False
|
|
if not await router.rename(conversation_id, name):
|
|
return False
|
|
|
|
store = _session_store()
|
|
if store is not None:
|
|
session_key = _session_key_for(conversation_id)
|
|
|
|
def _rename_route() -> None:
|
|
with store._lock:
|
|
store._ensure_loaded_locked()
|
|
entry = store._entries.get(session_key)
|
|
if entry is not None:
|
|
entry.display_name = name
|
|
store._save()
|
|
|
|
try:
|
|
await asyncio.to_thread(_rename_route)
|
|
except Exception:
|
|
logger.debug("[pheby] Hermes display-name update failed",
|
|
exc_info=True)
|
|
return True
|
|
|
|
|
|
async def delete_conversation(conversation_id: str) -> bool:
|
|
"""Delete a transcript and remove its live routing entry.
|
|
|
|
Hermes currently has no public per-key removal method. We therefore use
|
|
the same lock/save discipline as SessionStore's own pruning code. Calling
|
|
``reset_session`` here would create a replacement entry and make the
|
|
deleted conversation immediately reappear.
|
|
"""
|
|
server = _current_server()
|
|
router = server.router if server else None
|
|
if router is None or active_run(conversation_id) is not None:
|
|
return False
|
|
|
|
store = _session_store()
|
|
db = _session_db()
|
|
session_key = _session_key_for(conversation_id)
|
|
session_id = None
|
|
if store is not None:
|
|
try:
|
|
session_id = await asyncio.to_thread(
|
|
store.peek_session_id, session_key)
|
|
except Exception:
|
|
session_id = None
|
|
router_known = await router.get_name(conversation_id) is not None
|
|
if not router_known and not session_id:
|
|
return False
|
|
|
|
if session_id and db is not None:
|
|
try:
|
|
deleted = await asyncio.to_thread(db.delete_session, session_id)
|
|
if deleted is False:
|
|
return False
|
|
except Exception:
|
|
logger.error("[pheby] session db delete failed", exc_info=True)
|
|
return False
|
|
|
|
if store is not None:
|
|
def _remove_route() -> None:
|
|
with store._lock:
|
|
store._ensure_loaded_locked()
|
|
if store._entries.pop(session_key, None) is not None:
|
|
store._save()
|
|
|
|
try:
|
|
await asyncio.to_thread(_remove_route)
|
|
except Exception:
|
|
logger.error("[pheby] routing removal failed", exc_info=True)
|
|
return False
|
|
|
|
await router.forget(conversation_id)
|
|
with _RUN_LOCK:
|
|
_ACTIVE_RUNS.pop(conversation_id, None)
|
|
_TOOL_EVENTS.pop(conversation_id, None)
|
|
return True
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
# Chat runs
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
async def send_chat(server: Any, conversation_id: str, text: str,
|
|
client: Any, request_id: Optional[str],
|
|
attachment_ids: Optional[List[str]] = None) -> None:
|
|
"""Deliver a user message into the Hermes gateway for this conversation.
|
|
|
|
The gateway's full pipeline (auth, sessions, tools, approvals, clarify,
|
|
deliverables, streaming) runs on the adapter's message handler. Pheby
|
|
adds nothing to the agent loop.
|
|
|
|
*attachment_ids* (optional) reference inbound uploads already registered
|
|
in ``server.store``; they are anchored to the user message, handed to
|
|
Hermes as ``media_urls``/``media_types`` (tool-accessible local files),
|
|
and small text files are additionally inlined into the message text.
|
|
"""
|
|
adapter = _current_adapter()
|
|
if adapter is None or not hasattr(adapter, "handle_message"):
|
|
await client.send_json(proto.error_event(
|
|
proto.ERR_INTERNAL, "Gateway not connected yet", request_id))
|
|
return
|
|
|
|
# Register the conversation so it survives restarts.
|
|
await server.router.ensure_conversation(conversation_id)
|
|
|
|
run_id = uuid.uuid4().hex[:16]
|
|
source = _source_for(conversation_id)
|
|
message_id = uuid.uuid4().hex[:12]
|
|
|
|
# ── inbound attachments ────────────────────────────────────────────────
|
|
media_urls: List[str] = []
|
|
media_types: List[str] = []
|
|
media_text_inlined: List[bool] = []
|
|
inline_blocks: List[str] = []
|
|
attachment_descs: List[Dict[str, Any]] = []
|
|
for aid in attachment_ids or []:
|
|
desc = server.store.describe(aid)
|
|
if desc is None or desc.get("conversation_id") != conversation_id or \
|
|
desc.get("direction") != "inbound" or desc.get("message_id"):
|
|
await client.send_json(proto.error_event(
|
|
proto.ERR_NOT_FOUND,
|
|
"Unknown or already sent attachment for this conversation", request_id))
|
|
return
|
|
attachment_descs.append(desc)
|
|
for desc in attachment_descs:
|
|
aid = desc["attachment_id"]
|
|
blob = server.store.get_blob_path(aid)
|
|
if blob is None:
|
|
await client.send_json(proto.error_event(
|
|
proto.ERR_NOT_FOUND, "Attachment unavailable", request_id))
|
|
return
|
|
media_urls.append(str(blob))
|
|
mime = str(desc.get("mime_type", "application/octet-stream"))
|
|
media_types.append(mime)
|
|
# Small text files become part of the message text so the model
|
|
# reads them directly without a tool round-trip.
|
|
fname = str(desc.get("filename", "file"))
|
|
inlined = False
|
|
if mime.startswith("text/") and int(desc.get("size", 0)) <= \
|
|
proto.INLINE_TEXT_BYTES:
|
|
try:
|
|
content = blob.read_text(encoding="utf-8", errors="replace")
|
|
inline_blocks.append(
|
|
f"Attached file: {fname}\n```{fname.rsplit('.', 1)[-1]}\n"
|
|
f"{content}\n```")
|
|
inlined = True
|
|
except OSError:
|
|
pass
|
|
media_text_inlined.append(inlined)
|
|
|
|
effective_text = text
|
|
if attachment_descs:
|
|
effective_text = f"[Pheby user message]\n{text}\n[/Pheby user message]"
|
|
if inline_blocks:
|
|
effective_text += "\n\n" + "\n\n".join(inline_blocks)
|
|
|
|
from gateway.platforms.base import MessageEvent, MessageType
|
|
event = MessageEvent(
|
|
text=effective_text,
|
|
message_type=(MessageType.PHOTO if media_types and
|
|
all(t.startswith("image/") for t in media_types)
|
|
else (MessageType.DOCUMENT if media_types
|
|
else MessageType.TEXT)),
|
|
source=source,
|
|
message_id=message_id,
|
|
metadata={"pheby_run_id": run_id},
|
|
media_urls=media_urls,
|
|
media_types=media_types,
|
|
media_text_inlined=media_text_inlined,
|
|
)
|
|
|
|
with _RUN_LOCK:
|
|
already_active = conversation_id in _ACTIVE_RUNS
|
|
if not already_active:
|
|
_ACTIVE_RUNS[conversation_id] = {
|
|
"run_id": run_id,
|
|
"started": asyncio.get_running_loop().time(),
|
|
}
|
|
_TOOL_EVENTS[conversation_id] = {}
|
|
if already_active:
|
|
await client.send_json(proto.error_event(
|
|
proto.ERR_RUN_ACTIVE,
|
|
"A run is already active for this conversation", request_id))
|
|
return
|
|
await client.send_json({
|
|
"type": proto.S_RUN_ACCEPTED,
|
|
"conversation_id": conversation_id,
|
|
"run_id": run_id,
|
|
**({"request_id": request_id} if request_id else {}),
|
|
})
|
|
|
|
draft_message_id = f"draft-{run_id}"
|
|
await server.broadcast({
|
|
"type": proto.S_MESSAGE_START,
|
|
"conversation_id": conversation_id,
|
|
"run_id": run_id,
|
|
"message_id": draft_message_id,
|
|
})
|
|
# Track the draft in the adapter so send()/edit_message() associate the
|
|
# final text with the announced draft message id.
|
|
if getattr(adapter, "_drafts", None) is not None:
|
|
adapter._drafts.setdefault(conversation_id, {
|
|
"message_id": draft_message_id, "text": ""})
|
|
|
|
# The base adapter's handle_message() spawns background tasks and
|
|
# returns quickly; the eventual reply arrives through adapter.send().
|
|
try:
|
|
await adapter.handle_message(event)
|
|
for desc in attachment_descs:
|
|
aid = desc["attachment_id"]
|
|
server.store.anchor_message(aid, message_id)
|
|
await server.broadcast({
|
|
"type": proto.S_ATTACHMENT_ADDED,
|
|
"conversation_id": conversation_id,
|
|
"attachment": server.store.describe(aid),
|
|
})
|
|
except Exception:
|
|
with _RUN_LOCK:
|
|
current = _ACTIVE_RUNS.get(conversation_id)
|
|
if current and current.get("run_id") == run_id:
|
|
_ACTIVE_RUNS.pop(conversation_id, None)
|
|
if getattr(adapter, "_drafts", None) is not None:
|
|
adapter._drafts.pop(conversation_id, None)
|
|
await server.broadcast({
|
|
"type": proto.S_RUN_FINISHED,
|
|
"conversation_id": conversation_id,
|
|
"run_id": run_id,
|
|
"status": "failed",
|
|
"error": "Gateway rejected the message",
|
|
})
|
|
raise
|
|
|
|
|
|
def active_run(conversation_id: str) -> Optional[Dict[str, Any]]:
|
|
with _RUN_LOCK:
|
|
run = _ACTIVE_RUNS.get(conversation_id)
|
|
return dict(run) if run else None
|
|
|
|
|
|
def active_run_id(conversation_id: str) -> Optional[str]:
|
|
run = active_run(conversation_id)
|
|
return str(run["run_id"]) if run else None
|
|
|
|
|
|
def record_tool_event(conversation_id: str, event: Dict[str, Any]) -> None:
|
|
tool_id = str(event.get("tool_call_id") or "")
|
|
if not tool_id:
|
|
return
|
|
with _RUN_LOCK:
|
|
bucket = _TOOL_EVENTS.setdefault(conversation_id, {})
|
|
bucket[tool_id] = dict(event)
|
|
if len(bucket) > 200:
|
|
bucket.pop(next(iter(bucket)))
|
|
|
|
|
|
def runtime_snapshot(conversation_id: str) -> Dict[str, Any]:
|
|
"""Recoverable transient state included by ``conversation.open``."""
|
|
session_key = _session_key_for(conversation_id)
|
|
with _RUN_LOCK:
|
|
run = _ACTIVE_RUNS.get(conversation_id)
|
|
tools = list(_TOOL_EVENTS.get(conversation_id, {}).values())
|
|
approvals = [dict(p["event"]) for p in _PENDING_APPROVALS.values()
|
|
if p.get("session_key") == session_key]
|
|
clarifications = [dict(p["event"]) for p in _PENDING_CLARIFIES.values()
|
|
if p.get("session_key") == session_key]
|
|
return {
|
|
"run": dict(run) if run else None,
|
|
"tools": tools,
|
|
"approvals": approvals,
|
|
"clarifications": clarifications,
|
|
}
|
|
|
|
|
|
def note_run_finished(conversation_id: str, status: str = "completed",
|
|
error: Optional[str] = None) -> None:
|
|
"""Called by the adapter when a turn completes/fails."""
|
|
with _RUN_LOCK:
|
|
run = _ACTIVE_RUNS.pop(conversation_id, None)
|
|
if run is None:
|
|
return
|
|
run_id = run["run_id"]
|
|
server = _current_server()
|
|
if server is None:
|
|
return
|
|
payload = {
|
|
"type": proto.S_RUN_FINISHED,
|
|
"conversation_id": conversation_id,
|
|
"status": status,
|
|
}
|
|
if run_id:
|
|
payload["run_id"] = run_id
|
|
if error:
|
|
payload["error"] = proto.safe_str(error, 300)
|
|
try:
|
|
adapter = _current_adapter()
|
|
if adapter is not None:
|
|
adapter.schedule_broadcast(payload)
|
|
except RuntimeError:
|
|
pass
|
|
|
|
|
|
async def cancel_run(conversation_id: str, run_id: Optional[str]) -> bool:
|
|
"""Cancel an active run via Hermes's supported interrupt path."""
|
|
adapter = _current_adapter()
|
|
run = active_run(conversation_id)
|
|
if run is None:
|
|
return False
|
|
if run_id and run["run_id"] != run_id:
|
|
return False # stale run id — nothing to cancel
|
|
session_key = _session_key_for(conversation_id)
|
|
|
|
runner = _runner()
|
|
interrupted = False
|
|
if runner is not None:
|
|
# Preferred: gateway's own /stop dispatch (cancels task + drains).
|
|
running = getattr(runner, "_running_agents", {}).get(session_key)
|
|
agent = running if running is not None else None
|
|
interrupt = getattr(agent, "interrupt", None)
|
|
if callable(interrupt):
|
|
try:
|
|
interrupt("Cancelled by Pheby client")
|
|
invalidate = getattr(
|
|
runner, "_invalidate_session_run_generation", None)
|
|
if callable(invalidate):
|
|
invalidate(session_key, reason="pheby_cancel")
|
|
interrupted = True
|
|
except Exception:
|
|
logger.debug("[pheby] agent interrupt failed", exc_info=True)
|
|
if not interrupted and adapter is not None:
|
|
try:
|
|
event = getattr(adapter, "_active_sessions", {}).get(session_key)
|
|
if event is not None:
|
|
await adapter.interrupt_session_activity(
|
|
session_key, conversation_id)
|
|
interrupted = True
|
|
except Exception:
|
|
logger.debug("[pheby] adapter interrupt failed", exc_info=True)
|
|
if interrupted:
|
|
with _RUN_LOCK:
|
|
_ACTIVE_RUNS.pop(conversation_id, None)
|
|
return interrupted
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
# Approvals
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
async def push_approval(approval_data: Dict[str, Any],
|
|
session_key: str) -> None:
|
|
"""Adapter callback: a dangerous action needs a human decision."""
|
|
approval_id = uuid.uuid4().hex[:12]
|
|
# gateway.run redacts the command before calling send_exec_approval.
|
|
command = approval_data.get("command", "")
|
|
choices: List[str] = ["once", "deny"]
|
|
if approval_data.get("allow_session", True):
|
|
choices.insert(1, "session")
|
|
if approval_data.get("allow_permanent", True):
|
|
choices.insert(-1, "always")
|
|
conversation_id = _conversation_from_session_key(session_key)
|
|
event: Dict[str, Any] = {
|
|
"type": proto.S_APPROVAL_REQUEST,
|
|
"approval_id": approval_id,
|
|
"session_key": session_key,
|
|
"conversation_id": conversation_id,
|
|
"run_id": active_run_id(conversation_id),
|
|
"command": proto.safe_str(command, 2000),
|
|
"description": proto.safe_str(
|
|
approval_data.get("description", ""), 1000),
|
|
"choices": choices,
|
|
"ts": proto.now_iso(),
|
|
}
|
|
_PENDING_APPROVALS[approval_id] = {
|
|
"session_key": session_key,
|
|
"created": time.monotonic(),
|
|
"event": event,
|
|
}
|
|
server = _current_server()
|
|
if server is not None:
|
|
await server.broadcast(event)
|
|
|
|
|
|
async def resolve_approval(approval_id: str, choice: str,
|
|
reason: Optional[str]) -> bool:
|
|
"""Forward an approval decision to Hermes (tools.approval primitives)."""
|
|
if choice not in ("once", "session", "always", "deny"):
|
|
return False
|
|
pending = _PENDING_APPROVALS.get(approval_id)
|
|
if pending is None:
|
|
return False
|
|
try:
|
|
from tools.approval import resolve_gateway_approval
|
|
count = await asyncio.to_thread(
|
|
resolve_gateway_approval,
|
|
pending["session_key"], choice, False, reason)
|
|
ok = count > 0
|
|
except Exception:
|
|
logger.error("[pheby] approval resolve failed", exc_info=True)
|
|
ok = False
|
|
if ok:
|
|
_PENDING_APPROVALS.pop(approval_id, None)
|
|
return ok
|
|
|
|
|
|
def fail_stale_approvals(max_age: float = 3600.0) -> None:
|
|
"""Drop approval IDs whose Hermes-side gate has surely timed out."""
|
|
now = time.monotonic()
|
|
for aid in [a for a, p in _PENDING_APPROVALS.items()
|
|
if now - p["created"] > max_age]:
|
|
_PENDING_APPROVALS.pop(aid, None)
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
# Clarifications
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
async def push_clarify(clarify_id: str, session_key: str, question: str,
|
|
choices: Optional[List[str]]) -> None:
|
|
"""Adapter callback: the agent needs the user to choose."""
|
|
conversation_id = _conversation_from_session_key(session_key)
|
|
event: Dict[str, Any] = {
|
|
"type": proto.S_CLARIFY_REQUEST,
|
|
"clarify_id": clarify_id,
|
|
"session_key": session_key,
|
|
"conversation_id": conversation_id,
|
|
"run_id": active_run_id(conversation_id),
|
|
"question": proto.safe_str(question, 2000),
|
|
"choices": [proto.safe_str(c, 300) for c in choices]
|
|
if choices else None,
|
|
"allow_free_text": True, # Hermes clarify always permits "Other"
|
|
"ts": proto.now_iso(),
|
|
}
|
|
_PENDING_CLARIFIES[clarify_id] = {
|
|
"session_key": session_key,
|
|
"created": time.monotonic(),
|
|
"event": event,
|
|
}
|
|
server = _current_server()
|
|
if server is not None:
|
|
await server.broadcast(event)
|
|
|
|
|
|
async def resolve_clarify(clarify_id: str, response: str) -> bool:
|
|
"""Forward a clarification answer to Hermes's clarify primitive."""
|
|
pending = _PENDING_CLARIFIES.get(clarify_id)
|
|
if pending is None:
|
|
return False
|
|
try:
|
|
from tools.clarify_gateway import resolve_gateway_clarify
|
|
ok = await asyncio.to_thread(
|
|
resolve_gateway_clarify, clarify_id, response)
|
|
if not ok:
|
|
# Might be an awaiting-text open-ended clarify: route via the
|
|
# session text path instead.
|
|
from tools.clarify_gateway import \
|
|
resolve_text_response_for_session
|
|
ok = await asyncio.to_thread(
|
|
resolve_text_response_for_session,
|
|
pending["session_key"], response)
|
|
except Exception:
|
|
logger.error("[pheby] clarify resolve failed", exc_info=True)
|
|
ok = False
|
|
if ok:
|
|
_PENDING_CLARIFIES.pop(clarify_id, None)
|
|
return bool(ok)
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
# Conversation-scoped YOLO / approval bypass
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
async def yolo_snapshot(conversation_id: str) -> Dict[str, Any]:
|
|
"""Return the effective dangerous-command approval bypass for one chat."""
|
|
store = _session_store()
|
|
session_key = _session_key_for(conversation_id)
|
|
session_id = None
|
|
if store is not None:
|
|
try:
|
|
session_id = await asyncio.to_thread(store.peek_session_id, session_key)
|
|
except Exception:
|
|
logger.debug("[pheby] yolo session lookup failed", exc_info=True)
|
|
if not session_id:
|
|
return {"ok": False, "code": proto.ERR_CONVERSATION_NOT_FOUND,
|
|
"message": "Conversation not found"}
|
|
|
|
from tools.approval import enable_session_yolo, is_session_yolo_enabled
|
|
enabled = is_session_yolo_enabled(session_key)
|
|
if not enabled:
|
|
db = _session_db()
|
|
if db is not None:
|
|
try:
|
|
meta = await asyncio.to_thread(db.get_session, str(session_id))
|
|
from hermes_state import SessionDB
|
|
if SessionDB.session_yolo_enabled(meta):
|
|
enable_session_yolo(session_key)
|
|
enabled = True
|
|
except Exception:
|
|
logger.debug("[pheby] persisted yolo read failed", exc_info=True)
|
|
return {"ok": True, "enabled": bool(enabled), "scope": "conversation",
|
|
"conversation_id": conversation_id}
|
|
|
|
|
|
async def set_yolo(conversation_id: str, enabled: bool) -> Dict[str, Any]:
|
|
"""Set YOLO for one chat and persist it with that Hermes session."""
|
|
store = _session_store()
|
|
session_key = _session_key_for(conversation_id)
|
|
session_id = None
|
|
if store is not None:
|
|
try:
|
|
session_id = await asyncio.to_thread(store.peek_session_id, session_key)
|
|
except Exception:
|
|
logger.debug("[pheby] yolo session lookup failed", exc_info=True)
|
|
if not session_id:
|
|
return {"ok": False, "code": proto.ERR_CONVERSATION_NOT_FOUND,
|
|
"message": "Conversation not found"}
|
|
|
|
from tools.approval import disable_session_yolo, enable_session_yolo
|
|
(enable_session_yolo if enabled else disable_session_yolo)(session_key)
|
|
db = _session_db()
|
|
if db is not None:
|
|
try:
|
|
await asyncio.to_thread(db.set_session_yolo, str(session_id), bool(enabled))
|
|
except Exception:
|
|
logger.warning("[pheby] yolo persistence failed", exc_info=True)
|
|
return {"ok": True, "enabled": bool(enabled), "scope": "conversation",
|
|
"conversation_id": conversation_id}
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
# Models & reasoning
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
async def models_snapshot(conversation_id: Optional[str] = None) -> Dict[str, Any]:
|
|
"""Providers + models Hermes currently exposes (credential-aware)."""
|
|
def _collect() -> Dict[str, Any]:
|
|
from hermes_cli.config import get_compatible_custom_providers
|
|
from hermes_cli.model_switch_providers import list_picker_providers
|
|
cfg = _load_cfg()
|
|
model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {}
|
|
current_model = str(model_cfg.get("default", "") or "")
|
|
current_provider = str(model_cfg.get("provider", "openrouter") or "")
|
|
custom_providers = get_compatible_custom_providers(cfg)
|
|
excluded = (cfg.get("model_catalog") or {}).get(
|
|
"excluded_providers", [])
|
|
providers = list_picker_providers(
|
|
current_provider=current_provider,
|
|
current_base_url=str(model_cfg.get("base_url", "") or ""),
|
|
current_model=current_model,
|
|
user_providers=cfg.get("providers") if isinstance(cfg, dict) else None,
|
|
custom_providers=custom_providers,
|
|
excluded_providers=excluded if isinstance(excluded, list) else [],
|
|
)
|
|
return {"providers": providers, "current_model": current_model,
|
|
"current_provider": current_provider}
|
|
try:
|
|
data = await asyncio.to_thread(_collect)
|
|
except Exception:
|
|
logger.error("[pheby] model listing failed", exc_info=True)
|
|
data = {"providers": [], "current_model": "", "current_provider": "",
|
|
"error": "Model catalog unavailable"}
|
|
data["supported_reasoning_efforts"] = list(proto.REASONING_EFFORTS)
|
|
if conversation_id:
|
|
current = await current_model_snapshot(conversation_id)
|
|
data["current_model"] = current.get("model", data["current_model"])
|
|
data["current_provider"] = current.get(
|
|
"provider", data["current_provider"])
|
|
data["scope"] = current.get("scope", "global")
|
|
data["ts"] = proto.now_iso()
|
|
return data
|
|
|
|
|
|
async def current_model_snapshot(
|
|
conversation_id: Optional[str] = None) -> Dict[str, Any]:
|
|
def _collect() -> Dict[str, Any]:
|
|
cfg = _load_cfg()
|
|
model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {}
|
|
return {"model": str(model_cfg.get("default", "") or ""),
|
|
"provider": str(model_cfg.get("provider", "") or "")}
|
|
try:
|
|
data = await asyncio.to_thread(_collect)
|
|
except Exception:
|
|
data = {"model": "", "provider": "", "error": "Config unavailable"}
|
|
data["scope"] = "global"
|
|
store = _session_store()
|
|
if conversation_id and store is not None:
|
|
try:
|
|
override = await asyncio.to_thread(
|
|
store.get_model_override, _session_key_for(conversation_id))
|
|
if override:
|
|
data["model"] = override.get("model", data["model"])
|
|
data["provider"] = override.get("provider", data["provider"])
|
|
data["scope"] = "conversation"
|
|
except Exception:
|
|
logger.debug("[pheby] model override read failed", exc_info=True)
|
|
data["conversation_id"] = conversation_id
|
|
data["ts"] = proto.now_iso()
|
|
return data
|
|
|
|
|
|
async def set_model(model: str, provider: Optional[str],
|
|
conversation_id: Optional[str]) -> Dict[str, Any]:
|
|
"""Change the active model via Hermes's session/global override path."""
|
|
if not model:
|
|
return {"ok": False, "code": proto.ERR_BAD_REQUEST,
|
|
"message": "model is required"}
|
|
home_token = None
|
|
secret_token = None
|
|
try:
|
|
from agent.secret_scope import (
|
|
build_profile_secret_scope, reset_secret_scope, set_secret_scope,
|
|
)
|
|
from hermes_cli.config import get_compatible_custom_providers
|
|
from hermes_cli.model_switch import switch_model
|
|
from hermes_constants import (
|
|
get_hermes_home, reset_hermes_home_override,
|
|
set_hermes_home_override,
|
|
)
|
|
|
|
adapter = _current_adapter()
|
|
profile_home = Path(
|
|
getattr(adapter, "hermes_home", None) or get_hermes_home())
|
|
# model.set is a WebSocket control request, not a gateway message turn,
|
|
# so the runner has not installed this profile's ContextVars for us.
|
|
# Bind them explicitly before load_config()/switch_model(); to_thread
|
|
# copies the current context into its worker.
|
|
home_token = set_hermes_home_override(str(profile_home))
|
|
secrets = await asyncio.to_thread(
|
|
build_profile_secret_scope, profile_home)
|
|
secret_token = set_secret_scope(secrets)
|
|
|
|
cfg = _load_cfg()
|
|
model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {}
|
|
result = await asyncio.to_thread(
|
|
switch_model,
|
|
model,
|
|
str(model_cfg.get("provider", "openrouter") or "openrouter"),
|
|
str(model_cfg.get("default", "") or ""),
|
|
str(model_cfg.get("base_url", "") or ""),
|
|
"", # current_api_key — scoped runtime resolution handles credentials
|
|
False, # is_global → session-scoped when conversation given
|
|
provider or "",
|
|
cfg.get("providers") if isinstance(cfg, dict) else None,
|
|
get_compatible_custom_providers(cfg),
|
|
)
|
|
except Exception as exc:
|
|
logger.error("[pheby] switch_model failed", exc_info=True)
|
|
return {"ok": False, "code": proto.ERR_BAD_REQUEST,
|
|
"message": proto.safe_str(exc, 200)}
|
|
finally:
|
|
if secret_token is not None:
|
|
reset_secret_scope(secret_token)
|
|
if home_token is not None:
|
|
reset_hermes_home_override(home_token)
|
|
|
|
ok = bool(getattr(result, "success", False))
|
|
if not ok:
|
|
error = (getattr(result, "error_message", "") or
|
|
getattr(result, "error", ""))
|
|
return {"ok": False, "code": proto.ERR_BAD_REQUEST,
|
|
"message": proto.safe_str(error, 300)}
|
|
|
|
resolved_model = (getattr(result, "new_model", "") or
|
|
getattr(result, "model", "") or model)
|
|
resolved_provider = (getattr(result, "target_provider", "") or
|
|
getattr(result, "provider", "") or provider or "")
|
|
override = {"model": resolved_model}
|
|
if resolved_provider:
|
|
override["provider"] = resolved_provider
|
|
|
|
store = _session_store()
|
|
if conversation_id and store is not None:
|
|
try:
|
|
session_key = _session_key_for(conversation_id)
|
|
if await asyncio.to_thread(
|
|
store.peek_session_id, session_key) is None:
|
|
return {"ok": False, "code": proto.ERR_CONVERSATION_NOT_FOUND,
|
|
"message": "Conversation not found"}
|
|
await asyncio.to_thread(store.set_model_override,
|
|
session_key, override)
|
|
|
|
# Match Hermes's native /model commit path: the persisted override
|
|
# survives restarts, while the richer in-memory override gives the
|
|
# very next turn its resolved endpoint/key/capabilities. Evict any
|
|
# cached agent so it cannot answer once more with the old model.
|
|
runner = _runner()
|
|
runtime_overrides = getattr(
|
|
runner, "_session_model_overrides", None)
|
|
if isinstance(runtime_overrides, dict):
|
|
runtime_overrides[session_key] = {
|
|
"model": resolved_model,
|
|
"provider": resolved_provider,
|
|
"api_key": getattr(result, "api_key", "") or "",
|
|
"base_url": getattr(result, "base_url", "") or "",
|
|
"api_mode": getattr(result, "api_mode", "") or "",
|
|
"request_overrides": dict(
|
|
getattr(result, "request_overrides", None) or {}),
|
|
"capabilities": dict(
|
|
getattr(result, "runtime_capabilities", None) or {}),
|
|
}
|
|
evict = getattr(runner, "_evict_cached_agent", None)
|
|
if callable(evict):
|
|
evict(session_key)
|
|
return {"ok": True, "model": resolved_model,
|
|
"provider": resolved_provider, "scope": "conversation"}
|
|
except Exception:
|
|
logger.debug("[pheby] session model override failed",
|
|
exc_info=True)
|
|
# Global fallback: persist via Hermes config save (same path /model
|
|
# --global uses).
|
|
try:
|
|
await asyncio.to_thread(_save_global_model, resolved_model,
|
|
resolved_provider)
|
|
return {"ok": True, "model": resolved_model,
|
|
"provider": resolved_provider, "scope": "global"}
|
|
except Exception as exc:
|
|
logger.error("[pheby] global model save failed", exc_info=True)
|
|
return {"ok": False, "code": proto.ERR_INTERNAL,
|
|
"message": proto.safe_str(exc, 200)}
|
|
|
|
|
|
def _save_global_model(model: str, provider: str) -> None:
|
|
from hermes_cli.config import save_config_value
|
|
save_config_value("model.default", model)
|
|
if provider:
|
|
save_config_value("model.provider", provider)
|
|
|
|
|
|
async def reasoning_snapshot(
|
|
conversation_id: Optional[str] = None) -> Dict[str, Any]:
|
|
def _collect() -> Dict[str, Any]:
|
|
from hermes_constants import resolve_reasoning_config
|
|
cfg = _load_cfg()
|
|
model_cfg = (cfg.get("model") or {}) if isinstance(cfg, dict) else {}
|
|
resolved = resolve_reasoning_config(
|
|
cfg, str(model_cfg.get("default", "") or ""))
|
|
if resolved is None:
|
|
return {"effort": None, "enabled": None}
|
|
if resolved.get("enabled") is False:
|
|
return {"effort": "none", "enabled": False}
|
|
return {"effort": resolved.get("effort"), "enabled": True}
|
|
try:
|
|
data = await asyncio.to_thread(_collect)
|
|
except Exception:
|
|
data = {"effort": None, "enabled": None,
|
|
"error": "Config unavailable"}
|
|
data["scope"] = "global"
|
|
runner = _runner()
|
|
if conversation_id and runner is not None:
|
|
try:
|
|
cfg = await asyncio.to_thread(
|
|
runner._resolve_session_reasoning_config,
|
|
session_key=_session_key_for(conversation_id), model="")
|
|
if cfg is not None:
|
|
data = ({"effort": "none", "enabled": False}
|
|
if cfg.get("enabled") is False else
|
|
{"effort": cfg.get("effort"), "enabled": True})
|
|
data["scope"] = "conversation"
|
|
except Exception:
|
|
logger.debug("[pheby] reasoning override read failed",
|
|
exc_info=True)
|
|
data["conversation_id"] = conversation_id
|
|
data["supported_efforts"] = ["none"] + list(proto.REASONING_EFFORTS)
|
|
data["ts"] = proto.now_iso()
|
|
return data
|
|
|
|
|
|
async def set_reasoning(effort: str,
|
|
conversation_id: Optional[str]) -> Dict[str, Any]:
|
|
"""Set reasoning effort (Hermes levels + 'none' to disable)."""
|
|
if effort not in ("none",) + proto.REASONING_EFFORTS:
|
|
return {"ok": False, "code": proto.ERR_BAD_REQUEST,
|
|
"message": f"effort must be one of: none, "
|
|
f"{', '.join(proto.REASONING_EFFORTS)}"}
|
|
parsed = {"enabled": False} if effort == "none" else {
|
|
"enabled": True, "effort": effort}
|
|
runner = _runner()
|
|
if runner is not None and conversation_id:
|
|
try:
|
|
store = _session_store()
|
|
if store is None or await asyncio.to_thread(
|
|
store.peek_session_id,
|
|
_session_key_for(conversation_id)) is None:
|
|
return {"ok": False, "code": proto.ERR_CONVERSATION_NOT_FOUND,
|
|
"message": "Conversation not found"}
|
|
await asyncio.to_thread(
|
|
runner._set_session_reasoning_override,
|
|
_session_key_for(conversation_id), parsed)
|
|
return {"ok": True, "effort": effort, "scope": "conversation"}
|
|
except Exception:
|
|
logger.debug("[pheby] session reasoning override failed",
|
|
exc_info=True)
|
|
try:
|
|
await asyncio.to_thread(_save_global_reasoning, effort)
|
|
return {"ok": True, "effort": effort, "scope": "global"}
|
|
except Exception as exc:
|
|
return {"ok": False, "code": proto.ERR_INTERNAL,
|
|
"message": proto.safe_str(exc, 200)}
|
|
|
|
|
|
def _save_global_reasoning(effort: str) -> None:
|
|
from hermes_cli.config import save_config_value
|
|
save_config_value("agent.reasoning_effort",
|
|
False if effort == "none" else effort)
|
|
|
|
|
|
def _load_cfg() -> Dict[str, Any]:
|
|
from hermes_cli.config import load_config
|
|
return load_config() or {}
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
# Process service references
|
|
# ═══════════════════════════════════════════════════════════════════════════
|
|
def set_server(server: Any) -> None:
|
|
global _SERVER
|
|
with _SERVICE_LOCK:
|
|
_SERVER = server
|
|
|
|
|
|
def _current_server() -> Any:
|
|
with _SERVICE_LOCK:
|
|
return _SERVER
|
|
|
|
|
|
__all__ = [
|
|
"set_adapter", "set_server", "clear_services", "create_conversation",
|
|
"list_conversations", "conversation_history", "rename_conversation",
|
|
"delete_conversation", "send_chat", "cancel_run", "note_run_finished",
|
|
"active_run", "active_run_id", "record_tool_event", "runtime_snapshot",
|
|
"push_approval", "resolve_approval", "fail_stale_approvals",
|
|
"push_clarify", "resolve_clarify", "models_snapshot",
|
|
"current_model_snapshot", "set_model", "reasoning_snapshot",
|
|
"set_reasoning",
|
|
]
|