Compare commits
10
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
69f4291892 | ||
|
|
57177544ff | ||
|
|
ea7981eba7 | ||
|
|
f1b8519670 | ||
|
|
f8fd30942c | ||
|
|
1967c590ed | ||
|
|
702f4df194 | ||
|
|
0092015496 | ||
|
|
9caa12f4ec | ||
|
|
4642762289 |
@@ -368,6 +368,71 @@ def _get_continuation_prompt(is_partial_stub: bool, dropped_tools: Optional[List
|
||||
)
|
||||
|
||||
|
||||
def _maybe_apply_session_routing(agent, user_message, conversation_history) -> None:
|
||||
"""Smart model routing at session start (cache-safe).
|
||||
|
||||
Fires at most once per agent, and only on the FIRST turn of a *fresh*
|
||||
session (empty ``conversation_history`` → no cached prefix to break).
|
||||
Picks a tier-appropriate model BEFORE the system prompt is built, then
|
||||
applies it via the same ``switch_model`` path ``/model`` uses (which
|
||||
nulls ``_cached_system_prompt`` so the prompt rebuilds for the new
|
||||
model). Everything is fail-open: any error leaves the agent untouched.
|
||||
"""
|
||||
if getattr(agent, "_smart_routing_applied", False):
|
||||
return
|
||||
# Only route a genuinely fresh session — never swap the model into a
|
||||
# resumed conversation, which would invalidate its cached history.
|
||||
if conversation_history:
|
||||
agent._smart_routing_applied = True
|
||||
return
|
||||
try:
|
||||
from agent import model_router
|
||||
|
||||
routing_cfg = model_router.get_routing_config()
|
||||
if not routing_cfg.get("enabled") or not routing_cfg.get("apply_to_sessions", True):
|
||||
agent._smart_routing_applied = True
|
||||
return
|
||||
|
||||
decision = model_router.route(
|
||||
user_message,
|
||||
current_model=getattr(agent, "model", "") or "",
|
||||
current_provider=getattr(agent, "provider", "") or "",
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("session routing: classification failed: %s", exc)
|
||||
agent._smart_routing_applied = True
|
||||
return
|
||||
|
||||
# Mark applied regardless of outcome so we never re-classify this agent.
|
||||
agent._smart_routing_applied = True
|
||||
if decision is None:
|
||||
return
|
||||
|
||||
try:
|
||||
agent.switch_model(
|
||||
new_model=decision.model,
|
||||
new_provider=decision.provider,
|
||||
api_key=decision.api_key or "",
|
||||
base_url=decision.base_url or "",
|
||||
api_mode=decision.api_mode or "",
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.warning("session routing: switch_model failed (%s) — staying", exc)
|
||||
return
|
||||
|
||||
logger.info(
|
||||
"session routing: tier=%s → %s (%s)",
|
||||
decision.tier, decision.model, decision.provider,
|
||||
)
|
||||
if routing_cfg.get("announce", True) and not getattr(agent, "quiet_mode", False):
|
||||
try:
|
||||
agent._safe_print(
|
||||
f"\n🧭 Auto-routed to {decision.model} ({decision.tier} tier)"
|
||||
)
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
|
||||
|
||||
def run_conversation(
|
||||
agent,
|
||||
user_message: str,
|
||||
@@ -396,6 +461,12 @@ def run_conversation(
|
||||
Returns:
|
||||
Dict: Complete conversation result with final response and message history
|
||||
"""
|
||||
# ── Smart model routing (session start, cache-safe) ──
|
||||
# Runs BEFORE build_turn_context so the (model-specific) system prompt is
|
||||
# built for the routed model. No-op unless smart_model_routing.enabled and
|
||||
# this is the first turn of a fresh session. See _maybe_apply_session_routing.
|
||||
_maybe_apply_session_routing(agent, user_message, conversation_history)
|
||||
|
||||
# ── Per-turn setup (the prologue) ──
|
||||
# All once-per-turn setup — stdio guarding, retry-counter resets, user
|
||||
# message sanitization, todo/nudge hydration, system-prompt restore-or-
|
||||
|
||||
@@ -0,0 +1,329 @@
|
||||
"""Smart model routing — the cheap "picker" behind ``smart_model_routing``.
|
||||
|
||||
A lightweight classifier labels an incoming request's complexity tier
|
||||
(``light`` / ``standard`` / ``heavy``) and maps it to a tier-appropriate
|
||||
model. This mirrors the Cursor "Auto" idea — right-size the model to the
|
||||
task — while respecting Hermes' sacred per-conversation prompt cache.
|
||||
|
||||
**Nous Portal only.** Routing is a Nous Portal feature: every tier resolves
|
||||
to a model served by the Nous Portal (``provider: nous``), and the router
|
||||
only engages when the active model is itself on Nous Portal. If the current
|
||||
model is on any other provider the router is a strict no-op, so it never
|
||||
silently switches a non-Nous user onto Nous. The Portal already fronts the
|
||||
frontier models across vendors (``anthropic/…``, ``openai/…``,
|
||||
``google/…``, ``x-ai/…``), so a single Nous credential covers every tier.
|
||||
|
||||
The router is consulted ONLY at points where there is no cached prefix to
|
||||
invalidate:
|
||||
|
||||
* at the start of a *fresh* session, before the first API call
|
||||
(:func:`run_conversation` gates on empty ``conversation_history``), and
|
||||
* at each ``delegate_task`` boundary, where subagents get fresh context.
|
||||
|
||||
It never swaps the main model mid-conversation — that is ``/model``'s job
|
||||
and it deliberately resets the cache.
|
||||
|
||||
Everything here fails open: a broken/slow/misconfigured classifier must
|
||||
never wedge a turn. On any failure the caller stays on the current model.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Ordered cheapest/smallest → most capable. Order is load-bearing: the
|
||||
# ``min_tier`` floor and tier comparisons rely on it.
|
||||
TIERS: Tuple[str, ...] = ("light", "standard", "heavy")
|
||||
|
||||
# Smart routing is a Nous Portal feature — every tier resolves through this
|
||||
# provider, and routing only engages when the active model is on it too.
|
||||
NOUS_PROVIDER = "nous"
|
||||
|
||||
|
||||
def _is_nous_provider(provider: str) -> bool:
|
||||
"""True when ``provider`` names the Nous Portal.
|
||||
|
||||
The router is Nous-only, so this gates both the active-model check (only
|
||||
route a session/parent already on Nous) and is the implied provider for
|
||||
every tier target.
|
||||
"""
|
||||
return (provider or "").strip().lower() == NOUS_PROVIDER
|
||||
|
||||
_CLASSIFIER_SYSTEM_PROMPT = (
|
||||
"You are a routing classifier for an autonomous AI coding agent. Read the "
|
||||
"user's request and label how much model capability it needs, as exactly "
|
||||
"one of these tiers:\n"
|
||||
"- light: trivial or quick — simple questions, tiny edits, lookups, "
|
||||
"formatting, one-line answers.\n"
|
||||
"- standard: ordinary coding and analysis — implement a function, explain "
|
||||
"code, write a normal test, routine debugging.\n"
|
||||
"- heavy: hard or sprawling — multi-file refactors, architecture/design, "
|
||||
"subtle debugging, deep multi-step reasoning, security-sensitive work.\n"
|
||||
"Bias toward the HIGHER tier when unsure; quality matters more than saving "
|
||||
"a little money. Respond with ONLY the single tier word, nothing else."
|
||||
)
|
||||
|
||||
# Cap the message we send to the classifier — the opening request can be huge
|
||||
# (pasted logs, files). The first ~4k chars carry the intent.
|
||||
_MAX_CLASSIFY_CHARS = 4000
|
||||
|
||||
|
||||
@dataclass
|
||||
class RoutingDecision:
|
||||
"""A resolved decision to run on a specific model.
|
||||
|
||||
``base_url`` / ``api_key`` / ``api_mode`` are resolved credentials ready
|
||||
to hand to ``AIAgent.switch_model`` (session routing) or to
|
||||
``_build_child_agent`` overrides (delegation routing).
|
||||
"""
|
||||
|
||||
tier: str
|
||||
provider: str
|
||||
model: str
|
||||
base_url: Optional[str]
|
||||
api_key: Optional[str]
|
||||
api_mode: Optional[str]
|
||||
reason: str = ""
|
||||
|
||||
|
||||
def get_routing_config(config: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
||||
"""Return the ``smart_model_routing`` config dict (never None)."""
|
||||
if config is None:
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
|
||||
config = load_config()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("model_router: load_config failed: %s", exc)
|
||||
return {}
|
||||
cfg = config.get("smart_model_routing") if isinstance(config, dict) else None
|
||||
return cfg if isinstance(cfg, dict) else {}
|
||||
|
||||
|
||||
def is_enabled(config: Optional[Dict[str, Any]] = None) -> bool:
|
||||
return bool(get_routing_config(config).get("enabled"))
|
||||
|
||||
|
||||
def _tier_index(tier: str) -> int:
|
||||
try:
|
||||
return TIERS.index(tier)
|
||||
except ValueError:
|
||||
return TIERS.index("standard")
|
||||
|
||||
|
||||
def _apply_min_tier_floor(tier: str, routing_cfg: Dict[str, Any]) -> str:
|
||||
"""Bump ``tier`` up to ``min_tier`` when a floor is configured."""
|
||||
floor = str(routing_cfg.get("min_tier") or "").strip().lower()
|
||||
if floor in TIERS and _tier_index(tier) < _tier_index(floor):
|
||||
return floor
|
||||
return tier
|
||||
|
||||
|
||||
def _parse_tier(raw: str, default_tier: str) -> str:
|
||||
"""Extract a tier word from a classifier response. Fail-open to default."""
|
||||
text = (raw or "").strip().lower()
|
||||
if not text:
|
||||
return default_tier
|
||||
# Exact single-word answer (the happy path) or first tier word mentioned.
|
||||
for tier in TIERS:
|
||||
if tier in text:
|
||||
return tier
|
||||
return default_tier
|
||||
|
||||
|
||||
def classify_complexity(
|
||||
message: str,
|
||||
*,
|
||||
routing_cfg: Optional[Dict[str, Any]] = None,
|
||||
timeout: float = 20.0,
|
||||
) -> Tuple[str, str]:
|
||||
"""Classify ``message`` into a complexity tier.
|
||||
|
||||
Returns ``(tier, reason)``. Always returns a valid tier — on any failure
|
||||
it returns the configured ``default_tier`` with a diagnostic reason.
|
||||
"""
|
||||
routing_cfg = routing_cfg if routing_cfg is not None else get_routing_config()
|
||||
default_tier = str(routing_cfg.get("default_tier") or "standard").strip().lower()
|
||||
if default_tier not in TIERS:
|
||||
default_tier = "standard"
|
||||
|
||||
if not (message or "").strip():
|
||||
return default_tier, "empty message"
|
||||
|
||||
try:
|
||||
from agent.auxiliary_client import (
|
||||
get_auxiliary_extra_body,
|
||||
get_text_auxiliary_client,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("model_router: auxiliary client import failed: %s", exc)
|
||||
return default_tier, "auxiliary client unavailable"
|
||||
|
||||
try:
|
||||
client, model = get_text_auxiliary_client("routing_classifier")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("model_router: get_text_auxiliary_client failed: %s", exc)
|
||||
return default_tier, "auxiliary client unavailable"
|
||||
|
||||
if client is None or not model:
|
||||
return default_tier, "no auxiliary client configured"
|
||||
|
||||
snippet = message.strip()[:_MAX_CLASSIFY_CHARS]
|
||||
try:
|
||||
resp = client.chat.completions.create(
|
||||
model=model,
|
||||
messages=[
|
||||
{"role": "system", "content": _CLASSIFIER_SYSTEM_PROMPT},
|
||||
{"role": "user", "content": snippet},
|
||||
],
|
||||
temperature=0,
|
||||
max_tokens=16,
|
||||
timeout=timeout,
|
||||
extra_body=get_auxiliary_extra_body() or None,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.info(
|
||||
"model_router: classifier call failed (%s) — using default tier %r",
|
||||
type(exc).__name__,
|
||||
default_tier,
|
||||
)
|
||||
return default_tier, f"classifier error: {type(exc).__name__}"
|
||||
|
||||
try:
|
||||
raw = resp.choices[0].message.content or ""
|
||||
except Exception: # noqa: BLE001
|
||||
raw = ""
|
||||
|
||||
tier = _parse_tier(raw, default_tier)
|
||||
logger.info("model_router: classified tier=%s (raw=%r)", tier, (raw or "")[:40])
|
||||
return tier, "classified"
|
||||
|
||||
|
||||
def _tier_model(tier: str, routing_cfg: Dict[str, Any]) -> str:
|
||||
"""Return the configured Nous Portal model for a tier ('' when unset).
|
||||
|
||||
A tier maps to a bare Nous model id (``"anthropic/claude-opus-4.8"``).
|
||||
For backward compatibility a ``{"model": "..."}`` dict is also accepted;
|
||||
any ``provider`` key is ignored — tiers always run on the Nous Portal.
|
||||
An empty value means "stay on the current/parent model" for that tier.
|
||||
"""
|
||||
tiers = routing_cfg.get("tiers")
|
||||
if not isinstance(tiers, dict):
|
||||
return ""
|
||||
entry = tiers.get(tier)
|
||||
if isinstance(entry, dict):
|
||||
entry = entry.get("model")
|
||||
return str(entry or "").strip()
|
||||
|
||||
|
||||
def _resolve_tier_credentials(provider: str, model: str) -> Optional[Dict[str, Any]]:
|
||||
"""Resolve full credentials for a tier's Nous Portal model.
|
||||
|
||||
``provider`` is always :data:`NOUS_PROVIDER` — tiers are Nous-only. Reuses
|
||||
the same runtime-provider resolver delegation uses, so a routed tier
|
||||
behaves identically to ``delegation.provider``/``model``. Returns None
|
||||
(fail-open) when Nous credentials can't be resolved.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.runtime_provider import resolve_runtime_provider
|
||||
|
||||
runtime = resolve_runtime_provider(requested=provider, target_model=model)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.warning(
|
||||
"model_router: cannot resolve tier provider %r (model %r): %s — "
|
||||
"staying on current model",
|
||||
provider,
|
||||
model,
|
||||
exc,
|
||||
)
|
||||
return None
|
||||
|
||||
api_key = runtime.get("api_key", "")
|
||||
if not api_key:
|
||||
logger.warning(
|
||||
"model_router: tier provider %r resolved but has no API key — "
|
||||
"staying on current model",
|
||||
provider,
|
||||
)
|
||||
return None
|
||||
|
||||
return {
|
||||
"provider": runtime.get("provider") or provider,
|
||||
"model": model or runtime.get("model") or "",
|
||||
"base_url": runtime.get("base_url"),
|
||||
"api_key": api_key,
|
||||
"api_mode": runtime.get("api_mode"),
|
||||
}
|
||||
|
||||
|
||||
def route(
|
||||
message: str,
|
||||
*,
|
||||
current_model: str,
|
||||
current_provider: str,
|
||||
config: Optional[Dict[str, Any]] = None,
|
||||
timeout: Optional[float] = None,
|
||||
) -> Optional[RoutingDecision]:
|
||||
"""Decide which model ``message`` should run on.
|
||||
|
||||
Returns a :class:`RoutingDecision` when the request should run on a
|
||||
*different* Nous Portal model than the current one, or ``None`` to stay
|
||||
put (routing disabled, not on Nous, tier unconfigured, no-op, or any
|
||||
resolution failure). ``None`` is the cache-safe outcome — the caller
|
||||
makes no change.
|
||||
"""
|
||||
routing_cfg = get_routing_config(config)
|
||||
if not routing_cfg.get("enabled"):
|
||||
return None
|
||||
|
||||
# Nous Portal only: never route a session/parent that isn't already on
|
||||
# Nous, so the feature can't silently move a user onto another provider.
|
||||
if not _is_nous_provider(current_provider):
|
||||
logger.debug(
|
||||
"model_router: current provider %r is not Nous Portal — staying",
|
||||
current_provider,
|
||||
)
|
||||
return None
|
||||
|
||||
if timeout is None:
|
||||
try:
|
||||
timeout = float(
|
||||
(config or {}).get("auxiliary", {})
|
||||
.get("routing_classifier", {})
|
||||
.get("timeout", 20)
|
||||
)
|
||||
except Exception: # noqa: BLE001
|
||||
timeout = 20.0
|
||||
|
||||
tier, reason = classify_complexity(message, routing_cfg=routing_cfg, timeout=timeout)
|
||||
tier = _apply_min_tier_floor(tier, routing_cfg)
|
||||
|
||||
model = _tier_model(tier, routing_cfg)
|
||||
if not model:
|
||||
# Tier intentionally maps to "stay on the current/parent model".
|
||||
logger.debug("model_router: tier %s has no target — staying", tier)
|
||||
return None
|
||||
|
||||
# Current provider is already known to be Nous (gated above), so a model
|
||||
# match alone means we're on the right tier — never break the cache.
|
||||
if model == (current_model or "").strip():
|
||||
logger.debug("model_router: tier %s already active (%s) — no-op", tier, model)
|
||||
return None
|
||||
|
||||
creds = _resolve_tier_credentials(NOUS_PROVIDER, model)
|
||||
if creds is None:
|
||||
return None
|
||||
|
||||
return RoutingDecision(
|
||||
tier=tier,
|
||||
provider=creds["provider"],
|
||||
model=creds["model"],
|
||||
base_url=creds["base_url"],
|
||||
api_key=creds["api_key"],
|
||||
api_mode=creds["api_mode"],
|
||||
reason=reason,
|
||||
)
|
||||
@@ -1 +0,0 @@
|
||||
/home/ben/nous/hermes-agent/.worktrees/hermes-e14f8918/apps/desktop/node_modules
|
||||
@@ -91,7 +91,6 @@ import {
|
||||
} from '@/store/session'
|
||||
|
||||
import { type AppView, ARTIFACTS_ROUTE, MESSAGING_ROUTE, SKILLS_ROUTE } from '../../routes'
|
||||
import { useWorkspaceGitRepos } from '../../session/hooks/use-workspace-git'
|
||||
import { SidebarPanelLabel } from '../../shell/sidebar-label'
|
||||
import type { SidebarNavItem } from '../../types'
|
||||
|
||||
@@ -298,8 +297,6 @@ interface ChatSidebarProps extends React.ComponentProps<typeof Sidebar> {
|
||||
onDeleteSession: (sessionId: string) => void
|
||||
onArchiveSession: (sessionId: string) => void
|
||||
onNewSessionInWorkspace: (path: null | string) => void
|
||||
onNewSessionWorktree: (path: null | string) => void
|
||||
requestGateway: <T = unknown>(method: string, params?: Record<string, unknown>) => Promise<T>
|
||||
onManageCronJob: (jobId: string) => void
|
||||
onTriggerCronJob: (jobId: string) => void
|
||||
}
|
||||
@@ -314,8 +311,6 @@ export function ChatSidebar({
|
||||
onDeleteSession,
|
||||
onArchiveSession,
|
||||
onNewSessionInWorkspace,
|
||||
onNewSessionWorktree,
|
||||
requestGateway,
|
||||
onManageCronJob,
|
||||
onTriggerCronJob
|
||||
}: ChatSidebarProps) {
|
||||
@@ -665,21 +660,6 @@ export function ChatSidebar({
|
||||
sessionProfileTotals
|
||||
])
|
||||
|
||||
// Probe each distinct workspace path for git-repo-ness (memoized, once per
|
||||
// path) so the per-group "new session in a worktree" fork icon only appears
|
||||
// for real repos.
|
||||
const workspacePaths = useMemo(
|
||||
() => agentGroups.map(g => g.path).filter((p): p is string => Boolean(p)),
|
||||
[agentGroups]
|
||||
)
|
||||
|
||||
const gitRepoPaths = useWorkspaceGitRepos(workspacePaths, requestGateway)
|
||||
|
||||
const agentGroupsWithRepo = useMemo(
|
||||
() => agentGroups.map(g => ({ ...g, isGitRepo: g.path ? gitRepoPaths.has(g.path) : false })),
|
||||
[agentGroups, gitRepoPaths]
|
||||
)
|
||||
|
||||
const displayAgentSessions = agentSessions
|
||||
|
||||
// Pagination is scope-aware. In "All profiles" mode it tracks the global
|
||||
@@ -701,7 +681,7 @@ export function ChatSidebar({
|
||||
|
||||
const recentsMeta = countLabel(agentSessions.length, knownSessionTotal)
|
||||
|
||||
const displayAgentGroups = showAllProfiles ? profileGroups : agentsGrouped ? agentGroupsWithRepo : undefined
|
||||
const displayAgentGroups = showAllProfiles ? profileGroups : agentsGrouped ? agentGroups : undefined
|
||||
|
||||
// The recents list owns its own (virtualized) scroll container only when it's a
|
||||
// long flat list. In that case it must keep its scroller even in short mode, so
|
||||
@@ -980,7 +960,6 @@ export function ChatSidebar({
|
||||
onArchiveSession={onArchiveSession}
|
||||
onDeleteSession={onDeleteSession}
|
||||
onNewSessionInWorkspace={showAllProfiles ? undefined : onNewSessionInWorkspace}
|
||||
onNewSessionWorktree={showAllProfiles ? undefined : onNewSessionWorktree}
|
||||
onReorder={showAllProfiles ? undefined : handleAgentDragEnd}
|
||||
onResumeSession={onResumeSession}
|
||||
onToggle={() => setSidebarRecentsOpen(!agentsOpen)}
|
||||
@@ -1147,7 +1126,6 @@ interface SidebarSessionGroup {
|
||||
onLoadMore?: () => void
|
||||
sourceId?: string
|
||||
totalCount?: number
|
||||
isGitRepo?: boolean
|
||||
}
|
||||
|
||||
interface MessagingSection {
|
||||
@@ -1170,7 +1148,6 @@ interface SidebarSessionsSectionProps {
|
||||
onArchiveSession: (sessionId: string) => void
|
||||
onTogglePin: (sessionId: string) => void
|
||||
onNewSessionInWorkspace?: (path: null | string) => void
|
||||
onNewSessionWorktree?: (path: null | string) => void
|
||||
pinned: boolean
|
||||
rootClassName?: string
|
||||
contentClassName?: string
|
||||
@@ -1198,7 +1175,6 @@ function SidebarSessionsSection({
|
||||
onArchiveSession,
|
||||
onTogglePin,
|
||||
onNewSessionInWorkspace,
|
||||
onNewSessionWorktree,
|
||||
pinned,
|
||||
rootClassName,
|
||||
contentClassName,
|
||||
@@ -1273,7 +1249,6 @@ function SidebarSessionsSection({
|
||||
group={group}
|
||||
key={group.id}
|
||||
onNewSession={onNewSessionInWorkspace}
|
||||
onNewSessionWorktree={onNewSessionWorktree}
|
||||
renderRows={renderNestedSessionList}
|
||||
/>
|
||||
) : (
|
||||
@@ -1281,7 +1256,6 @@ function SidebarSessionsSection({
|
||||
group={group}
|
||||
key={group.id}
|
||||
onNewSession={onNewSessionInWorkspace}
|
||||
onNewSessionWorktree={onNewSessionWorktree}
|
||||
renderRows={renderSessionList}
|
||||
/>
|
||||
)
|
||||
@@ -1352,7 +1326,6 @@ interface SidebarWorkspaceGroupProps extends React.ComponentProps<'div'> {
|
||||
group: SidebarSessionGroup
|
||||
renderRows: (sessions: SessionInfo[]) => React.ReactNode
|
||||
onNewSession?: (path: null | string) => void
|
||||
onNewSessionWorktree?: (path: null | string) => void
|
||||
reorderable?: boolean
|
||||
dragging?: boolean
|
||||
dragHandleProps?: React.HTMLAttributes<HTMLElement>
|
||||
@@ -1362,7 +1335,6 @@ function SidebarWorkspaceGroup({
|
||||
group,
|
||||
renderRows,
|
||||
onNewSession,
|
||||
onNewSessionWorktree,
|
||||
reorderable = false,
|
||||
dragging = false,
|
||||
dragHandleProps,
|
||||
@@ -1454,17 +1426,6 @@ function SidebarWorkspaceGroup({
|
||||
</button>
|
||||
</Tip>
|
||||
)}
|
||||
{group.isGitRepo && onNewSessionWorktree && group.path && (
|
||||
<button
|
||||
aria-label={`New worktree session in ${group.label}`}
|
||||
className="grid size-4 shrink-0 place-items-center rounded-sm bg-transparent text-(--ui-text-quaternary) opacity-0 transition-opacity hover:bg-(--ui-control-hover-background) hover:text-foreground group-hover/workspace:opacity-100"
|
||||
onClick={() => onNewSessionWorktree(group.path)}
|
||||
title={`New session in a git worktree of ${group.label}`}
|
||||
type="button"
|
||||
>
|
||||
<Codicon name="repo-forked" size="0.75rem" />
|
||||
</button>
|
||||
)}
|
||||
{reorderable && (
|
||||
<span
|
||||
{...dragHandleProps}
|
||||
@@ -1515,7 +1476,6 @@ interface SortableWorkspaceProps {
|
||||
group: SidebarSessionGroup
|
||||
renderRows: (sessions: SessionInfo[]) => React.ReactNode
|
||||
onNewSession?: (path: null | string) => void
|
||||
onNewSessionWorktree?: (path: null | string) => void
|
||||
}
|
||||
|
||||
function SortableSidebarWorkspaceGroup(props: SortableWorkspaceProps) {
|
||||
|
||||
@@ -70,7 +70,6 @@ import {
|
||||
setMessagingPlatformTotals,
|
||||
setMessagingSessions,
|
||||
setMessagingTruncated,
|
||||
setPendingWorktree,
|
||||
setSessionProfileTotals,
|
||||
setSessions,
|
||||
setSessionsLoading,
|
||||
@@ -676,23 +675,6 @@ export function DesktopController() {
|
||||
[requestGateway, startFreshSessionDraft]
|
||||
)
|
||||
|
||||
const startSessionInWorktree = useCallback(
|
||||
(path: null | string) => {
|
||||
const target = path?.trim()
|
||||
|
||||
if (!target) {
|
||||
return
|
||||
}
|
||||
|
||||
// Same as a workspace new-session, but arm the one-shot worktree flag so
|
||||
// the backend creates the session inside a fresh git worktree of this
|
||||
// repo. startFreshSessionDraft() clears the flag, so arm it afterwards.
|
||||
startSessionInWorkspace(target)
|
||||
setPendingWorktree(true)
|
||||
},
|
||||
[startSessionInWorkspace]
|
||||
)
|
||||
|
||||
const handleSkinCommand = useSkinCommand()
|
||||
|
||||
const {
|
||||
@@ -809,14 +791,12 @@ export function DesktopController() {
|
||||
}}
|
||||
onNavigate={selectSidebarItem}
|
||||
onNewSessionInWorkspace={startSessionInWorkspace}
|
||||
onNewSessionWorktree={startSessionInWorktree}
|
||||
onResumeSession={sessionId => navigate(sessionRoute(sessionId))}
|
||||
onTriggerCronJob={jobId => {
|
||||
void triggerCronJob(jobId)
|
||||
.then(() => refreshCronJobs())
|
||||
.catch(() => undefined)
|
||||
}}
|
||||
requestGateway={requestGateway}
|
||||
/>
|
||||
)
|
||||
|
||||
|
||||
@@ -17,9 +17,10 @@ import { $activeGatewayProfile, $newChatProfile, ensureGatewayProfile, normalize
|
||||
import {
|
||||
$currentCwd,
|
||||
$messages,
|
||||
$pendingWorktree,
|
||||
$sessions,
|
||||
$yoloActive,
|
||||
getRememberedWorkspaceCwd,
|
||||
workspaceCwdForNewSession,
|
||||
sessionPinId,
|
||||
setActiveSessionId,
|
||||
setAwaitingResponse,
|
||||
@@ -36,14 +37,12 @@ import {
|
||||
setFreshDraftReady,
|
||||
setIntroSeed,
|
||||
setMessages,
|
||||
setPendingWorktree,
|
||||
setSelectedStoredSessionId,
|
||||
setSessions,
|
||||
setSessionStartedAt,
|
||||
setSessionsTotal,
|
||||
setTurnStartedAt,
|
||||
setYoloActive,
|
||||
workspaceCwdForNewSession
|
||||
setYoloActive
|
||||
} from '@/store/session'
|
||||
import { reportBackendContract } from '@/store/updates'
|
||||
import type { SessionCreateResponse, SessionInfo, SessionResumeResponse, UsageStats } from '@/types/hermes'
|
||||
@@ -317,8 +316,6 @@ export function useSessionActions({
|
||||
// otherwise the sticky last-used workspace (PR #37586).
|
||||
setCurrentCwd(workspaceCwdForNewSession())
|
||||
setCurrentBranch('')
|
||||
// A plain new-chat draft is never a worktree session; clear any stale arm.
|
||||
setPendingWorktree(false)
|
||||
clearComposerDraft()
|
||||
clearComposerAttachments()
|
||||
setFreshDraftReady(true)
|
||||
@@ -342,21 +339,11 @@ export function useSessionActions({
|
||||
// Pass the owning profile so a new chat under a non-launch profile (global
|
||||
// remote mode) builds its agent + persists against THAT profile's home/db.
|
||||
const newChatProfile = $newChatProfile.get()
|
||||
// The fork icon arms a one-shot worktree request; consume + reset it so
|
||||
// a later plain new-chat doesn't accidentally inherit it.
|
||||
const worktree = $pendingWorktree.get()
|
||||
|
||||
if (worktree) {
|
||||
setPendingWorktree(false)
|
||||
}
|
||||
|
||||
const created = await requestGateway<SessionCreateResponse>('session.create', {
|
||||
cols: 96,
|
||||
...(cwd && { cwd }),
|
||||
...(newChatProfile ? { profile: newChatProfile } : {}),
|
||||
...(worktree && cwd ? { worktree: true } : {})
|
||||
...(newChatProfile ? { profile: newChatProfile } : {})
|
||||
})
|
||||
|
||||
const stored = created.stored_session_id ?? null
|
||||
|
||||
if (
|
||||
|
||||
@@ -1,65 +0,0 @@
|
||||
import { useEffect } from 'react'
|
||||
import { useStore } from '@nanostores/react'
|
||||
import { atom } from 'nanostores'
|
||||
|
||||
/**
|
||||
* Per-workspace git-repo detection for the sidebar.
|
||||
*
|
||||
* The "new session in a worktree" fork icon must only appear for workspace
|
||||
* groups whose path is a real git repository. We probe each distinct path once
|
||||
* via the `git.is_repo` gateway method and memoize the answer for the lifetime
|
||||
* of the renderer — a workspace doesn't stop being a repo while the app is open,
|
||||
* and re-probing on every sidebar render would be wasteful.
|
||||
*
|
||||
* Results live in a module-level nanostore so every sidebar instance shares one
|
||||
* cache and re-renders when a probe resolves.
|
||||
*/
|
||||
|
||||
// path -> isRepo. Absence means "not yet probed".
|
||||
const $repoByPath = atom<Record<string, boolean>>({})
|
||||
|
||||
// Paths with an in-flight or completed probe, so we never probe the same path
|
||||
// twice (even before the first result lands).
|
||||
const probed = new Set<string>()
|
||||
|
||||
type RequestGateway = <T = unknown>(method: string, params?: Record<string, unknown>) => Promise<T>
|
||||
|
||||
async function probePath(path: string, requestGateway: RequestGateway): Promise<void> {
|
||||
try {
|
||||
const res = await requestGateway<{ is_repo?: boolean }>('git.is_repo', { cwd: path })
|
||||
$repoByPath.set({ ...$repoByPath.get(), [path]: Boolean(res?.is_repo) })
|
||||
} catch {
|
||||
// Treat a failed probe as "not a repo" — the icon simply won't appear, and
|
||||
// the backend would fall back gracefully anyway if it somehow got asked.
|
||||
$repoByPath.set({ ...$repoByPath.get(), [path]: false })
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Probe every supplied workspace path for git-repo-ness (once each) and return
|
||||
* a `Set` of the paths that are repos. Re-renders when probes resolve.
|
||||
*
|
||||
* @param paths Distinct, non-null workspace paths to probe.
|
||||
* @param requestGateway Gateway RPC caller.
|
||||
*/
|
||||
export function useWorkspaceGitRepos(paths: string[], requestGateway: RequestGateway): Set<string> {
|
||||
const repoByPath = useStore($repoByPath)
|
||||
|
||||
useEffect(() => {
|
||||
for (const path of paths) {
|
||||
if (!path || probed.has(path)) {
|
||||
continue
|
||||
}
|
||||
probed.add(path)
|
||||
void probePath(path, requestGateway)
|
||||
}
|
||||
}, [paths, requestGateway])
|
||||
|
||||
const repos = new Set<string>()
|
||||
for (const [path, isRepo] of Object.entries(repoByPath)) {
|
||||
if (isRepo) {
|
||||
repos.add(path)
|
||||
}
|
||||
}
|
||||
return repos
|
||||
}
|
||||
@@ -186,10 +186,6 @@ export const $currentFastMode = atom(false)
|
||||
export const $yoloActive = atom(false)
|
||||
export const $currentCwd = atom(getRememberedWorkspaceCwd())
|
||||
export const $currentBranch = atom('')
|
||||
// When true, the next backend session is created inside a fresh git worktree of
|
||||
// $currentCwd (set by the sidebar's "new session in a worktree" fork icon).
|
||||
// Consumed and reset by the create-session path.
|
||||
export const $pendingWorktree = atom(false)
|
||||
export const $currentUsage = atom<UsageStats>({
|
||||
calls: 0,
|
||||
input: 0,
|
||||
@@ -244,7 +240,6 @@ export const workspaceCwdForNewSession = (): string =>
|
||||
getConfiguredDefaultProjectDir() || getRememberedWorkspaceCwd() || $currentCwd.get().trim()
|
||||
|
||||
export const setCurrentBranch = (next: Updater<string>) => updateAtom($currentBranch, next)
|
||||
export const setPendingWorktree = (next: Updater<boolean>) => updateAtom($pendingWorktree, next)
|
||||
export const setCurrentUsage = (next: Updater<UsageStats>) => updateAtom($currentUsage, next)
|
||||
export const setSessionStartedAt = (next: Updater<number | null>) => updateAtom($sessionStartedAt, next)
|
||||
export const setTurnStartedAt = (next: Updater<number | null>) => updateAtom($turnStartedAt, next)
|
||||
|
||||
@@ -890,6 +890,10 @@ def _cleanup_all_browsers(*args, **kwargs):
|
||||
|
||||
# Guard to prevent cleanup from running multiple times on exit
|
||||
_cleanup_done = False
|
||||
# One-shot CLI finalization runs before process cleanup so plugins can observe
|
||||
# the session boundary while the agent is still attached. If a signal lands in
|
||||
# that narrow window, atexit cleanup must not emit that session finalize again.
|
||||
_single_query_finalize_attempted_session_ids: set[str | None] = set()
|
||||
# Weak reference to the active AIAgent for memory provider shutdown at exit
|
||||
_active_agent_ref = None
|
||||
_deferred_agent_startup_done = False
|
||||
@@ -989,11 +993,13 @@ def _run_cleanup(*, notify_session_finalize: bool = True):
|
||||
# Shut down memory provider (on_session_end + shutdown_all) at actual
|
||||
# session boundary — NOT per-turn inside run_conversation().
|
||||
if notify_session_finalize:
|
||||
_notify_session_finalize(
|
||||
session_id=_active_agent_ref.session_id if _active_agent_ref else None,
|
||||
platform="cli",
|
||||
reason="shutdown",
|
||||
)
|
||||
cleanup_session_id = _active_agent_ref.session_id if _active_agent_ref else None
|
||||
if _should_emit_cleanup_session_finalize(cleanup_session_id):
|
||||
_notify_session_finalize(
|
||||
session_id=cleanup_session_id,
|
||||
platform="cli",
|
||||
reason="shutdown",
|
||||
)
|
||||
try:
|
||||
if _active_agent_ref and hasattr(_active_agent_ref, 'shutdown_memory_provider'):
|
||||
# Forward the agent's own transcript so memory providers'
|
||||
@@ -1011,6 +1017,14 @@ def _run_cleanup(*, notify_session_finalize: bool = True):
|
||||
pass
|
||||
|
||||
|
||||
def _should_emit_cleanup_session_finalize(session_id: str | None) -> bool:
|
||||
if not _single_query_finalize_attempted_session_ids:
|
||||
return True
|
||||
if session_id is None:
|
||||
return False
|
||||
return session_id not in _single_query_finalize_attempted_session_ids
|
||||
|
||||
|
||||
def _notify_session_finalize(
|
||||
*,
|
||||
session_id: str | None,
|
||||
@@ -1068,11 +1082,17 @@ def _emit_interrupted_session_end(cli, *, reason: str = "keyboard_interrupt") ->
|
||||
def _notify_single_query_session_finalize(cli, *, reason: str = "shutdown") -> None:
|
||||
agent = getattr(cli, "agent", None)
|
||||
session_id = getattr(agent, "session_id", None) or getattr(cli, "session_id", None)
|
||||
_notify_session_finalize(
|
||||
session_id=session_id,
|
||||
platform=getattr(agent, "platform", None) or "cli",
|
||||
reason=reason,
|
||||
)
|
||||
if session_id in _single_query_finalize_attempted_session_ids:
|
||||
return
|
||||
|
||||
try:
|
||||
_notify_session_finalize(
|
||||
session_id=session_id,
|
||||
platform=getattr(agent, "platform", None) or "cli",
|
||||
reason=reason,
|
||||
)
|
||||
finally:
|
||||
_single_query_finalize_attempted_session_ids.add(session_id)
|
||||
|
||||
|
||||
def _finalize_single_query(cli) -> None:
|
||||
|
||||
@@ -262,6 +262,14 @@ if [ -d "$HERMES_HOME/profiles" ]; then
|
||||
chown -R hermes:hermes "$HERMES_HOME/profiles" 2>/dev/null || true
|
||||
fi
|
||||
|
||||
# Always reset ownership of $HERMES_HOME/cron on every boot for the same
|
||||
# docker-exec/root-write reason as profiles/. The cron scheduler state
|
||||
# (jobs.json) must stay readable by the unprivileged hermes runtime even
|
||||
# after root-context maintenance commands or scheduler writes.
|
||||
if [ -d "$HERMES_HOME/cron" ]; then
|
||||
chown -R hermes:hermes "$HERMES_HOME/cron" 2>/dev/null || true
|
||||
fi
|
||||
|
||||
# Reset ownership of hermes-owned top-level state files on every boot.
|
||||
# The targeted data-volume chown above only covers hermes-owned
|
||||
# *subdirectories*; loose state files living directly under $HERMES_HOME
|
||||
|
||||
@@ -1341,6 +1341,20 @@ DEFAULT_CONFIG = {
|
||||
"timeout": 600,
|
||||
"extra_body": {},
|
||||
},
|
||||
# Routing classifier — the cheap "picker" that smart_model_routing
|
||||
# consults to label an incoming request's complexity tier (light /
|
||||
# standard / heavy). Runs on the Nous Portal (smart routing is
|
||||
# Nous-only). Point this at a small, fast Portal model: it runs once
|
||||
# per fresh session and per delegated subtask, so an expensive model
|
||||
# here defeats the purpose. "auto" = use the main chat model.
|
||||
"routing_classifier": {
|
||||
"provider": "nous",
|
||||
"model": "google/gemini-3.5-flash",
|
||||
"base_url": "",
|
||||
"api_key": "",
|
||||
"timeout": 20,
|
||||
"extra_body": {},
|
||||
},
|
||||
},
|
||||
|
||||
"display": {
|
||||
@@ -1718,6 +1732,46 @@ DEFAULT_CONFIG = {
|
||||
"subagent_auto_approve": False,
|
||||
},
|
||||
|
||||
# Smart model routing — a cheap "picker" classifies an incoming request's
|
||||
# complexity tier and routes it to a tier-appropriate model. Mirrors the
|
||||
# Cursor "Auto" idea (right-size the model to the task) while respecting
|
||||
# Hermes' sacred prompt-cache: routing only ever happens at points where
|
||||
# there is no cached prefix to invalidate — at the START of a fresh
|
||||
# session (before the first API call) and at each delegate_task boundary
|
||||
# (subagents get fresh context). It never swaps the main model mid-
|
||||
# conversation (that is what `/model` is for, and it resets the cache).
|
||||
#
|
||||
# Nous Portal only: every tier runs on the Nous Portal, and routing only
|
||||
# engages when the active model is itself on Nous Portal (otherwise it is
|
||||
# a strict no-op — it never moves a non-Nous user onto Nous). The Portal
|
||||
# fronts the frontier models across vendors, so one credential covers
|
||||
# every tier.
|
||||
#
|
||||
# Off by default. The classifier runs via auxiliary.routing_classifier —
|
||||
# point that at a cheap, fast Portal model (see its comment above).
|
||||
"smart_model_routing": {
|
||||
"enabled": False, # master switch
|
||||
"apply_to_sessions": True, # route at the start of a fresh session
|
||||
"apply_to_delegation": True, # route delegated subtasks by their goal
|
||||
# Which Nous Portal model each complexity tier runs on. Leave a tier
|
||||
# empty to "stay on the current/parent model" — the natural baseline
|
||||
# for `standard`. Credentials resolve automatically from the Nous
|
||||
# provider, exactly like delegation.model.
|
||||
"tiers": {
|
||||
"light": "google/gemini-3.5-flash", # fast + cheap
|
||||
"standard": "", # empty = main model
|
||||
"heavy": "anthropic/claude-opus-4.8", # frontier
|
||||
},
|
||||
# Tier used when the classifier is unreachable or returns garbage.
|
||||
# Fail-open: a broken picker must never wedge a turn.
|
||||
"default_tier": "standard",
|
||||
# Quality-first guardrail: never route below this tier. Set to
|
||||
# "standard" to forbid the "light" tier entirely. Empty = no floor.
|
||||
"min_tier": "",
|
||||
# Surface the routing decision to the user (tier + chosen model).
|
||||
"announce": True,
|
||||
},
|
||||
|
||||
# Ephemeral prefill messages file — JSON list of {role, content} dicts
|
||||
# injected at the start of every API call for few-shot priming.
|
||||
# Never saved to sessions, logs, or trajectories.
|
||||
|
||||
@@ -5557,58 +5557,6 @@ class SessionRename(BaseModel):
|
||||
profile: Optional[str] = None
|
||||
|
||||
|
||||
def _cleanup_session_worktree(db, sid: str) -> Optional[bool]:
|
||||
"""Remove the git worktree owned by a session, honoring the unpushed guard.
|
||||
|
||||
Returns:
|
||||
* ``None`` — the session had no worktree (nothing to do).
|
||||
* ``False`` — the worktree was removed.
|
||||
* ``True`` — the worktree was *preserved* because its branch has commits
|
||||
not reachable from any remote (the unpushed-commits guard in
|
||||
``_cleanup_worktree`` declined to delete it).
|
||||
|
||||
On removal, the worktree mapping is cleared from the session row so a repeat
|
||||
archive is a no-op. Best-effort: never raises into the request handler.
|
||||
"""
|
||||
try:
|
||||
row = db.get_session(sid)
|
||||
except Exception:
|
||||
return None
|
||||
if not row:
|
||||
return None
|
||||
wt_path = row.get("worktree_path")
|
||||
if not wt_path:
|
||||
return None
|
||||
|
||||
info = {
|
||||
"path": wt_path,
|
||||
"branch": row.get("worktree_branch"),
|
||||
"repo_root": row.get("worktree_repo_root"),
|
||||
}
|
||||
try:
|
||||
from cli import _cleanup_worktree
|
||||
except Exception:
|
||||
_log.debug("worktree cleanup helper unavailable", exc_info=True)
|
||||
return None
|
||||
|
||||
try:
|
||||
_cleanup_worktree(info)
|
||||
except Exception:
|
||||
_log.debug("worktree cleanup failed for session %s", sid, exc_info=True)
|
||||
return None
|
||||
|
||||
# _cleanup_worktree preserves worktrees with unpushed commits. Probe the
|
||||
# filesystem to learn what actually happened: gone → removed (clear the
|
||||
# mapping); still present → preserved (keep the mapping so we can retry).
|
||||
still_present = os.path.isdir(wt_path)
|
||||
if not still_present:
|
||||
try:
|
||||
db.set_session_worktree(sid, None)
|
||||
except Exception:
|
||||
_log.debug("failed to clear worktree mapping for %s", sid, exc_info=True)
|
||||
return still_present
|
||||
|
||||
|
||||
@app.patch("/api/sessions/{session_id}")
|
||||
async def rename_session_endpoint(session_id: str, body: SessionRename):
|
||||
"""Update a session: rename (or clear its title) and/or archive it.
|
||||
@@ -5638,14 +5586,6 @@ async def rename_session_endpoint(session_id: str, body: SessionRename):
|
||||
result = {"ok": True, "title": db.get_session_title(sid) or ""}
|
||||
if body.archived is not None:
|
||||
result["archived"] = bool(body.archived)
|
||||
# Archiving a session reclaims its git worktree (created via the
|
||||
# desktop "new session in a worktree" fork icon). _cleanup_worktree
|
||||
# keeps the unpushed-commits guard, so a worktree whose branch has
|
||||
# commits not on any remote is preserved rather than destroyed.
|
||||
if body.archived:
|
||||
preserved = _cleanup_session_worktree(db, sid)
|
||||
if preserved is not None:
|
||||
result["worktree_preserved"] = preserved
|
||||
return result
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
@@ -471,9 +471,6 @@ CREATE TABLE IF NOT EXISTS sessions (
|
||||
handoff_error TEXT,
|
||||
rewind_count INTEGER NOT NULL DEFAULT 0,
|
||||
archived INTEGER NOT NULL DEFAULT 0,
|
||||
worktree_path TEXT,
|
||||
worktree_branch TEXT,
|
||||
worktree_repo_root TEXT,
|
||||
FOREIGN KEY (parent_session_id) REFERENCES sessions(id)
|
||||
);
|
||||
|
||||
@@ -1730,33 +1727,6 @@ class SessionDB:
|
||||
rowcount = self._execute_write(_do)
|
||||
return rowcount > 0
|
||||
|
||||
def set_session_worktree(
|
||||
self,
|
||||
session_id: str,
|
||||
worktree_path: Optional[str],
|
||||
worktree_branch: Optional[str] = None,
|
||||
worktree_repo_root: Optional[str] = None,
|
||||
) -> bool:
|
||||
"""Record (or clear) the git worktree owned by a session.
|
||||
|
||||
Desktop "new session in a worktree" creates a throwaway git worktree and
|
||||
runs the session inside it. We stamp the worktree path/branch/repo-root
|
||||
on the row so the archive handler can find and remove the worktree later
|
||||
— even across a backend restart, when the in-memory session map is gone.
|
||||
|
||||
Pass ``worktree_path=None`` to clear the mapping (e.g. after cleanup).
|
||||
Returns True when a row was updated.
|
||||
"""
|
||||
def _do(conn):
|
||||
cursor = conn.execute(
|
||||
"UPDATE sessions SET worktree_path = ?, worktree_branch = ?, "
|
||||
"worktree_repo_root = ? WHERE id = ?",
|
||||
(worktree_path, worktree_branch, worktree_repo_root, session_id),
|
||||
)
|
||||
return cursor.rowcount
|
||||
rowcount = self._execute_write(_do)
|
||||
return rowcount > 0
|
||||
|
||||
def get_session_by_title(self, title: str) -> Optional[Dict[str, Any]]:
|
||||
"""Look up a session by exact title. Returns session dict or None."""
|
||||
with self._lock:
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
/home/ben/nous/hermes-agent/.worktrees/hermes-e14f8918/node_modules
|
||||
@@ -227,7 +227,30 @@ def _trace_key(task_id: str, session_id: str) -> str:
|
||||
return f"thread:{threading.get_ident()}"
|
||||
|
||||
|
||||
def _truncate_text(value: str, max_chars: int) -> str:
|
||||
def _is_base64_data_uri(value: str) -> bool:
|
||||
prefix = value[:200].lower()
|
||||
return prefix.startswith("data:") and ";base64," in prefix
|
||||
|
||||
|
||||
def _redact_data_uri(value: str) -> dict[str, Any]:
|
||||
header = value.split(",", 1)[0] if "," in value else "data:"
|
||||
media_type = header[5:].split(";", 1)[0] if header.startswith("data:") else ""
|
||||
return {
|
||||
"type": "data_uri",
|
||||
"media_type": media_type or None,
|
||||
"omitted": True,
|
||||
"length": len(value),
|
||||
}
|
||||
|
||||
|
||||
def _truncate_text(value: str, max_chars: int) -> Any:
|
||||
# Langfuse SDK treats data:*;base64 strings as media and attempts to
|
||||
# decode them. Truncating those strings produces invalid base64 and noisy
|
||||
# "Error parsing base64 data URI" logs. Observability only needs metadata,
|
||||
# not raw image/audio payloads, so redact the whole data URI before it
|
||||
# reaches the SDK.
|
||||
if _is_base64_data_uri(value):
|
||||
return _redact_data_uri(value)
|
||||
if len(value) <= max_chars:
|
||||
return value
|
||||
return value[:max_chars] + f"... [truncated {len(value) - max_chars} chars]"
|
||||
|
||||
@@ -130,6 +130,7 @@ AUTHOR_MAP = {
|
||||
"wangpuv@hotmail.com": "wangpuv",
|
||||
"202622897+ticketclosed-wontfix@users.noreply.github.com": "ticketclosed-wontfix",
|
||||
"wuxuebin1993@gmail.com": "victorGPT",
|
||||
"xiaoxingitee@gmail.com": "xiaoxinova",
|
||||
"wei.chen.coder@gmail.com": "wenchengxucool",
|
||||
"frowte3k@gmail.com": "Frowtek",
|
||||
"211828103+julio-cloudvisor@users.noreply.github.com": "julio-cloudvisor",
|
||||
|
||||
@@ -0,0 +1,329 @@
|
||||
"""Tests for smart model routing (agent/model_router.py + wiring).
|
||||
|
||||
All tests are hermetic — the classifier and credential resolution are
|
||||
stubbed, so nothing hits the network. The invariants under test:
|
||||
|
||||
* routing is a strict no-op when disabled (the default),
|
||||
* routing is Nous-only — a non-Nous session is never touched, and every
|
||||
tier resolves through the Nous provider,
|
||||
* a tier that maps to the current model never triggers a switch
|
||||
(the cache-safety guarantee),
|
||||
* the min_tier floor is honored,
|
||||
* classification fails open to the default tier,
|
||||
* explicit delegation/model pins beat routing,
|
||||
* the session-start helper fires at most once and skips resumed sessions.
|
||||
"""
|
||||
|
||||
import types
|
||||
|
||||
import pytest
|
||||
|
||||
from agent import model_router
|
||||
from agent.model_router import RoutingDecision
|
||||
|
||||
|
||||
def _cfg(**routing):
|
||||
# Tiers are Nous Portal model ids (smart routing is Nous-only).
|
||||
base = {
|
||||
"enabled": True,
|
||||
"apply_to_sessions": True,
|
||||
"apply_to_delegation": True,
|
||||
"tiers": {
|
||||
"light": "google/gemini-3.5-flash",
|
||||
"standard": "",
|
||||
"heavy": "anthropic/claude-opus-4.8",
|
||||
},
|
||||
"default_tier": "standard",
|
||||
"min_tier": "",
|
||||
"announce": True,
|
||||
}
|
||||
base.update(routing)
|
||||
return {"smart_model_routing": base}
|
||||
|
||||
|
||||
# ── pure helpers ────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_parse_tier_exact_and_embedded():
|
||||
assert model_router._parse_tier("heavy", "standard") == "heavy"
|
||||
assert model_router._parse_tier(" Light\n", "standard") == "light"
|
||||
assert model_router._parse_tier("I think this is standard work", "heavy") == "standard"
|
||||
|
||||
|
||||
def test_parse_tier_fails_open_to_default():
|
||||
assert model_router._parse_tier("", "standard") == "standard"
|
||||
assert model_router._parse_tier("banana", "heavy") == "heavy"
|
||||
|
||||
|
||||
def test_min_tier_floor_bumps_up():
|
||||
cfg = _cfg(min_tier="standard")["smart_model_routing"]
|
||||
assert model_router._apply_min_tier_floor("light", cfg) == "standard"
|
||||
assert model_router._apply_min_tier_floor("heavy", cfg) == "heavy"
|
||||
|
||||
|
||||
def test_min_tier_floor_ignores_invalid():
|
||||
cfg = _cfg(min_tier="bogus")["smart_model_routing"]
|
||||
assert model_router._apply_min_tier_floor("light", cfg) == "light"
|
||||
|
||||
|
||||
def test_tier_model_reads_config():
|
||||
cfg = _cfg()["smart_model_routing"]
|
||||
assert model_router._tier_model("light", cfg) == "google/gemini-3.5-flash"
|
||||
assert model_router._tier_model("standard", cfg) == ""
|
||||
|
||||
|
||||
def test_tier_model_accepts_legacy_dict_and_ignores_provider():
|
||||
cfg = _cfg(
|
||||
tiers={"heavy": {"provider": "anthropic", "model": "anthropic/claude-opus-4.8"}}
|
||||
)["smart_model_routing"]
|
||||
assert model_router._tier_model("heavy", cfg) == "anthropic/claude-opus-4.8"
|
||||
|
||||
|
||||
def test_is_nous_provider():
|
||||
assert model_router._is_nous_provider("nous")
|
||||
assert model_router._is_nous_provider(" Nous ")
|
||||
assert not model_router._is_nous_provider("openrouter")
|
||||
assert not model_router._is_nous_provider("")
|
||||
|
||||
|
||||
# ── route() behavior ──────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_route_disabled_is_noop():
|
||||
decision = model_router.route(
|
||||
"anything",
|
||||
current_model="openai/gpt-5.5",
|
||||
current_provider="nous",
|
||||
config=_cfg(enabled=False),
|
||||
)
|
||||
assert decision is None
|
||||
|
||||
|
||||
def test_route_noop_when_not_on_nous(monkeypatch):
|
||||
# Nous-only: an enabled router never touches a non-Nous session.
|
||||
monkeypatch.setattr(model_router, "classify_complexity", lambda *a, **k: ("heavy", "x"))
|
||||
decision = model_router.route(
|
||||
"hard refactor",
|
||||
current_model="gpt-5.4",
|
||||
current_provider="openrouter",
|
||||
config=_cfg(),
|
||||
)
|
||||
assert decision is None
|
||||
|
||||
|
||||
def test_route_tier_with_no_target_stays(monkeypatch):
|
||||
# standard tier maps to empty → stay on current model.
|
||||
monkeypatch.setattr(model_router, "classify_complexity", lambda *a, **k: ("standard", "x"))
|
||||
decision = model_router.route(
|
||||
"normal task",
|
||||
current_model="openai/gpt-5.5",
|
||||
current_provider="nous",
|
||||
config=_cfg(),
|
||||
)
|
||||
assert decision is None
|
||||
|
||||
|
||||
def test_route_noop_when_tier_matches_current(monkeypatch):
|
||||
# heavy tier resolves to the model we're already on → no switch (cache-safe).
|
||||
monkeypatch.setattr(model_router, "classify_complexity", lambda *a, **k: ("heavy", "x"))
|
||||
decision = model_router.route(
|
||||
"hard refactor",
|
||||
current_model="anthropic/claude-opus-4.8",
|
||||
current_provider="nous",
|
||||
config=_cfg(),
|
||||
)
|
||||
assert decision is None
|
||||
|
||||
|
||||
def test_route_returns_decision_on_tier_change(monkeypatch):
|
||||
monkeypatch.setattr(model_router, "classify_complexity", lambda *a, **k: ("heavy", "x"))
|
||||
monkeypatch.setattr(
|
||||
model_router,
|
||||
"_resolve_tier_credentials",
|
||||
lambda p, m: {"provider": "nous", "model": m,
|
||||
"base_url": "https://inference-api.nousresearch.com/v1",
|
||||
"api_key": "sk", "api_mode": None},
|
||||
)
|
||||
decision = model_router.route(
|
||||
"hard refactor",
|
||||
current_model="openai/gpt-5.5",
|
||||
current_provider="nous",
|
||||
config=_cfg(),
|
||||
)
|
||||
assert isinstance(decision, RoutingDecision)
|
||||
assert decision.tier == "heavy"
|
||||
assert decision.model == "anthropic/claude-opus-4.8"
|
||||
assert decision.provider == "nous"
|
||||
|
||||
|
||||
def test_route_resolves_tier_against_nous(monkeypatch):
|
||||
# The tier model is always resolved through the Nous provider.
|
||||
monkeypatch.setattr(model_router, "classify_complexity", lambda *a, **k: ("light", "x"))
|
||||
captured = {}
|
||||
|
||||
def _fake_resolve(provider, model):
|
||||
captured["provider"] = provider
|
||||
captured["model"] = model
|
||||
return {"provider": provider, "model": model, "base_url": None,
|
||||
"api_key": "sk", "api_mode": None}
|
||||
|
||||
monkeypatch.setattr(model_router, "_resolve_tier_credentials", _fake_resolve)
|
||||
decision = model_router.route(
|
||||
"tiny edit",
|
||||
current_model="openai/gpt-5.5",
|
||||
current_provider="nous",
|
||||
config=_cfg(),
|
||||
)
|
||||
assert decision is not None
|
||||
assert captured["provider"] == model_router.NOUS_PROVIDER
|
||||
assert captured["model"] == "google/gemini-3.5-flash"
|
||||
|
||||
|
||||
def test_route_fails_open_when_credentials_unresolved(monkeypatch):
|
||||
monkeypatch.setattr(model_router, "classify_complexity", lambda *a, **k: ("light", "x"))
|
||||
monkeypatch.setattr(model_router, "_resolve_tier_credentials", lambda p, m: None)
|
||||
decision = model_router.route(
|
||||
"tiny edit",
|
||||
current_model="openai/gpt-5.5",
|
||||
current_provider="nous",
|
||||
config=_cfg(),
|
||||
)
|
||||
assert decision is None
|
||||
|
||||
|
||||
def test_route_honors_min_tier(monkeypatch):
|
||||
# classifier says light, but min_tier=heavy forces heavy.
|
||||
monkeypatch.setattr(model_router, "classify_complexity", lambda *a, **k: ("light", "x"))
|
||||
captured = {}
|
||||
|
||||
def _fake_resolve(provider, model):
|
||||
captured["provider"] = provider
|
||||
captured["model"] = model
|
||||
return {"provider": provider, "model": model, "base_url": None,
|
||||
"api_key": "sk", "api_mode": None}
|
||||
|
||||
monkeypatch.setattr(model_router, "_resolve_tier_credentials", _fake_resolve)
|
||||
decision = model_router.route(
|
||||
"tiny edit",
|
||||
current_model="openai/gpt-5.5",
|
||||
current_provider="nous",
|
||||
config=_cfg(min_tier="heavy"),
|
||||
)
|
||||
assert decision is not None
|
||||
assert decision.tier == "heavy"
|
||||
assert captured["model"] == "anthropic/claude-opus-4.8"
|
||||
|
||||
|
||||
# ── classify_complexity fail-open ─────────────────────────────────────────
|
||||
|
||||
|
||||
def test_classify_fails_open_without_aux_client(monkeypatch):
|
||||
import agent.auxiliary_client as aux
|
||||
|
||||
monkeypatch.setattr(aux, "get_text_auxiliary_client", lambda task: (None, None))
|
||||
tier, reason = model_router.classify_complexity(
|
||||
"do something", routing_cfg=_cfg()["smart_model_routing"]
|
||||
)
|
||||
assert tier == "standard"
|
||||
assert "no auxiliary client" in reason
|
||||
|
||||
|
||||
def test_classify_empty_message_returns_default():
|
||||
tier, reason = model_router.classify_complexity(
|
||||
" ", routing_cfg=_cfg(default_tier="heavy")["smart_model_routing"]
|
||||
)
|
||||
assert tier == "heavy"
|
||||
|
||||
|
||||
# ── session-start wiring (_maybe_apply_session_routing) ───────────────────
|
||||
|
||||
|
||||
class _FakeAgent:
|
||||
def __init__(self):
|
||||
self.model = "openai/gpt-5.5"
|
||||
self.provider = "nous"
|
||||
self.quiet_mode = True
|
||||
self.switched = None
|
||||
self._smart_routing_applied = False
|
||||
|
||||
def switch_model(self, **kwargs):
|
||||
self.switched = kwargs
|
||||
self.model = kwargs["new_model"]
|
||||
self.provider = kwargs["new_provider"]
|
||||
|
||||
|
||||
def test_session_routing_skips_resumed_session(monkeypatch):
|
||||
from agent import conversation_loop
|
||||
|
||||
agent = _FakeAgent()
|
||||
# Non-empty history → must not classify or switch, but must mark applied.
|
||||
conversation_loop._maybe_apply_session_routing(agent, "hi", [{"role": "user", "content": "x"}])
|
||||
assert agent.switched is None
|
||||
assert agent._smart_routing_applied is True
|
||||
|
||||
|
||||
def test_session_routing_applies_once_and_switches(monkeypatch):
|
||||
from agent import conversation_loop
|
||||
|
||||
agent = _FakeAgent()
|
||||
monkeypatch.setattr(model_router, "get_routing_config", lambda config=None: _cfg()["smart_model_routing"])
|
||||
monkeypatch.setattr(
|
||||
model_router,
|
||||
"route",
|
||||
lambda *a, **k: RoutingDecision(
|
||||
tier="heavy", provider="nous", model="anthropic/claude-opus-4.8",
|
||||
base_url=None, api_key="sk", api_mode=None, reason="classified",
|
||||
),
|
||||
)
|
||||
conversation_loop._maybe_apply_session_routing(agent, "hard task", None)
|
||||
assert agent.switched is not None
|
||||
assert agent.model == "anthropic/claude-opus-4.8"
|
||||
assert agent._smart_routing_applied is True
|
||||
|
||||
# Second call must be a no-op (flag already set).
|
||||
agent.switched = None
|
||||
conversation_loop._maybe_apply_session_routing(agent, "another", None)
|
||||
assert agent.switched is None
|
||||
|
||||
|
||||
# ── delegation wiring (_route_task_creds) ─────────────────────────────────
|
||||
|
||||
|
||||
def test_delegation_routing_respects_explicit_model():
|
||||
from tools import delegate_tool
|
||||
|
||||
base = {"model": "pinned/model", "provider": "nous", "base_url": None,
|
||||
"api_key": None, "api_mode": None}
|
||||
parent = types.SimpleNamespace(model="openai/gpt-5.5", provider="nous")
|
||||
out = delegate_tool._route_task_creds(base, "anything", parent)
|
||||
assert out is base # unchanged — explicit delegation.model wins
|
||||
|
||||
|
||||
def test_delegation_routing_sets_model_when_unpinned(monkeypatch):
|
||||
from tools import delegate_tool
|
||||
|
||||
monkeypatch.setattr(model_router, "get_routing_config", lambda config=None: _cfg()["smart_model_routing"])
|
||||
monkeypatch.setattr(
|
||||
model_router,
|
||||
"route",
|
||||
lambda *a, **k: RoutingDecision(
|
||||
tier="light", provider="nous", model="google/gemini-3.5-flash",
|
||||
base_url=None, api_key="sk", api_mode=None, reason="classified",
|
||||
),
|
||||
)
|
||||
base = {"model": None, "provider": None, "base_url": None, "api_key": None, "api_mode": None}
|
||||
parent = types.SimpleNamespace(model="openai/gpt-5.5", provider="nous")
|
||||
out = delegate_tool._route_task_creds(base, "tiny task", parent)
|
||||
assert out["model"] == "google/gemini-3.5-flash"
|
||||
assert out["provider"] == "nous"
|
||||
|
||||
|
||||
def test_delegation_routing_noop_returns_base(monkeypatch):
|
||||
from tools import delegate_tool
|
||||
|
||||
monkeypatch.setattr(model_router, "get_routing_config", lambda config=None: _cfg()["smart_model_routing"])
|
||||
monkeypatch.setattr(model_router, "route", lambda *a, **k: None)
|
||||
base = {"model": None, "provider": None, "base_url": None, "api_key": None, "api_mode": None}
|
||||
parent = types.SimpleNamespace(model="openai/gpt-5.5", provider="nous")
|
||||
out = delegate_tool._route_task_creds(base, "task", parent)
|
||||
assert out is base
|
||||
@@ -5,6 +5,12 @@ import pytest
|
||||
import cli
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def reset_single_query_finalize_state(monkeypatch):
|
||||
monkeypatch.setattr(cli, "_single_query_finalize_attempted_session_ids", set())
|
||||
monkeypatch.setattr(cli, "_cleanup_done", False)
|
||||
|
||||
|
||||
def test_finalize_single_query_runs_cleanup_without_reemitting_finalize_before_release(monkeypatch):
|
||||
calls = []
|
||||
fake_cli = SimpleNamespace(_release_active_session=lambda: calls.append(("release", {})))
|
||||
@@ -70,6 +76,54 @@ def test_finalize_single_query_runs_cleanup_when_finalize_hook_fails(monkeypatch
|
||||
assert calls == ["finalize", "cleanup", "release"]
|
||||
|
||||
|
||||
def test_finalize_single_query_signal_window_does_not_reemit_during_atexit(monkeypatch):
|
||||
calls = []
|
||||
fake_agent = SimpleNamespace(session_id="agent-session", platform="cli")
|
||||
fake_cli = SimpleNamespace(
|
||||
agent=fake_agent,
|
||||
session_id="cli-session",
|
||||
_release_active_session=lambda: calls.append(("release", {})),
|
||||
)
|
||||
|
||||
def invoke_hook(name, **kwargs):
|
||||
calls.append((name, kwargs))
|
||||
|
||||
def interrupted_cleanup(**_kwargs):
|
||||
raise KeyboardInterrupt()
|
||||
|
||||
expected_finalize = (
|
||||
"on_session_finalize",
|
||||
{
|
||||
"session_id": "agent-session",
|
||||
"platform": "cli",
|
||||
"reason": "shutdown",
|
||||
},
|
||||
)
|
||||
|
||||
original_run_cleanup = cli._run_cleanup
|
||||
monkeypatch.setattr("hermes_cli.plugins.invoke_hook", invoke_hook)
|
||||
monkeypatch.setattr(cli, "_run_cleanup", interrupted_cleanup)
|
||||
|
||||
with pytest.raises(KeyboardInterrupt):
|
||||
cli._finalize_single_query(fake_cli)
|
||||
|
||||
assert calls == [expected_finalize, ("release", {})]
|
||||
|
||||
# Simulate later atexit cleanup after the interrupted one-shot path. The
|
||||
# active agent may already be unavailable by then.
|
||||
monkeypatch.setattr(cli, "_run_cleanup", original_run_cleanup)
|
||||
monkeypatch.setattr(cli, "_active_agent_ref", None)
|
||||
monkeypatch.setattr(cli, "_reset_terminal_input_modes_on_exit", lambda: None)
|
||||
monkeypatch.setattr(cli, "_cleanup_all_terminals", lambda: None)
|
||||
monkeypatch.setattr(cli, "_cleanup_all_browsers", lambda: None)
|
||||
monkeypatch.setattr("tools.mcp_tool.shutdown_mcp_servers", lambda: None)
|
||||
monkeypatch.setattr("agent.auxiliary_client.shutdown_cached_clients", lambda: None)
|
||||
|
||||
cli._run_cleanup()
|
||||
|
||||
assert calls == [expected_finalize, ("release", {})]
|
||||
|
||||
|
||||
def test_notify_single_query_session_finalize_uses_agent_session(monkeypatch):
|
||||
calls = []
|
||||
fake_agent = SimpleNamespace(session_id="agent-session", platform="cli")
|
||||
|
||||
@@ -171,6 +171,40 @@ class TestHooksInert:
|
||||
mod.on_post_tool_call(tool_name="read_file", args={}, result="ok", task_id="t", session_id="s")
|
||||
|
||||
|
||||
class TestPayloadSanitization:
|
||||
def test_safe_value_redacts_base64_data_uri_instead_of_truncating(self):
|
||||
sys.modules.pop("plugins.observability.langfuse", None)
|
||||
import importlib
|
||||
mod = importlib.import_module("plugins.observability.langfuse")
|
||||
|
||||
payload = "data:image/png;base64," + ("a" * 20000)
|
||||
result = mod._safe_value(payload)
|
||||
|
||||
assert result == {
|
||||
"type": "data_uri",
|
||||
"media_type": "image/png",
|
||||
"omitted": True,
|
||||
"length": len(payload),
|
||||
}
|
||||
|
||||
def test_serialize_messages_redacts_data_uri_parts(self):
|
||||
sys.modules.pop("plugins.observability.langfuse", None)
|
||||
import importlib
|
||||
mod = importlib.import_module("plugins.observability.langfuse")
|
||||
|
||||
payload = "data:image/jpeg;base64," + ("b" * 20000)
|
||||
serialized = mod._serialize_messages([
|
||||
{"role": "user", "content": [{"type": "image_url", "image_url": {"url": payload}}]}
|
||||
])
|
||||
|
||||
assert serialized[0]["content"][0]["image_url"]["url"] == {
|
||||
"type": "data_uri",
|
||||
"media_type": "image/jpeg",
|
||||
"omitted": True,
|
||||
"length": len(payload),
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Placeholder-credential guard (#23823).
|
||||
#
|
||||
|
||||
@@ -1,222 +0,0 @@
|
||||
"""Tests for the desktop "new session in a worktree" feature.
|
||||
|
||||
Covers the three backend pieces:
|
||||
* ``SessionDB.set_session_worktree`` persistence + read-back (hermes_state).
|
||||
* ``tui_gateway.server._git_is_repo`` / ``_create_session_worktree`` — worktree
|
||||
creation wired into ``session.create``, with eager DB-row persistence.
|
||||
* ``hermes_cli.web_server._cleanup_session_worktree`` — archive-time cleanup
|
||||
that honours the unpushed-commits guard.
|
||||
|
||||
The worktree machinery itself lives in ``cli.py`` and is exercised here against
|
||||
a real temporary git repo (no mocks) so the create/cleanup round-trip is proven
|
||||
end-to-end, not just in unit isolation.
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_state import SessionDB
|
||||
|
||||
|
||||
def _init_git_repo(path: str, *, with_commit: bool = True, with_remote: bool = False) -> None:
|
||||
"""Initialise a real git repo at *path* for worktree tests."""
|
||||
subprocess.run(["git", "init", "-q", path], check=True)
|
||||
subprocess.run(["git", "-C", path, "config", "user.email", "t@example.com"], check=True)
|
||||
subprocess.run(["git", "-C", path, "config", "user.name", "Tester"], check=True)
|
||||
if with_commit:
|
||||
readme = os.path.join(path, "README.md")
|
||||
with open(readme, "w", encoding="utf-8") as f:
|
||||
f.write("hello\n")
|
||||
subprocess.run(["git", "-C", path, "add", "."], check=True)
|
||||
subprocess.run(["git", "-C", path, "commit", "-qm", "init"], check=True)
|
||||
if with_remote:
|
||||
# A bare "remote" so unpushed-commit detection has a baseline to compare
|
||||
# against (without a remote, _worktree_has_unpushed_commits treats the
|
||||
# worktree as having nothing unpushed).
|
||||
remote = path + "-remote.git"
|
||||
subprocess.run(["git", "init", "-q", "--bare", remote], check=True)
|
||||
subprocess.run(["git", "-C", path, "remote", "add", "origin", remote], check=True)
|
||||
subprocess.run(["git", "-C", path, "push", "-q", "origin", "HEAD"], check=True)
|
||||
|
||||
|
||||
class TestSetSessionWorktree:
|
||||
def test_persists_and_reads_back(self, tmp_path):
|
||||
db = SessionDB(db_path=tmp_path / "state.db")
|
||||
try:
|
||||
db.create_session("s1", source="tui", cwd="/repo/.worktrees/hermes-abc")
|
||||
assert db.set_session_worktree(
|
||||
"s1", "/repo/.worktrees/hermes-abc", "hermes/hermes-abc", "/repo"
|
||||
)
|
||||
row = db.get_session("s1")
|
||||
assert row["worktree_path"] == "/repo/.worktrees/hermes-abc"
|
||||
assert row["worktree_branch"] == "hermes/hermes-abc"
|
||||
assert row["worktree_repo_root"] == "/repo"
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
def test_clear_mapping(self, tmp_path):
|
||||
db = SessionDB(db_path=tmp_path / "state.db")
|
||||
try:
|
||||
db.create_session("s1", source="tui")
|
||||
db.set_session_worktree("s1", "/wt", "hermes/x", "/repo")
|
||||
assert db.set_session_worktree("s1", None)
|
||||
row = db.get_session("s1")
|
||||
assert row["worktree_path"] is None
|
||||
assert row["worktree_branch"] is None
|
||||
assert row["worktree_repo_root"] is None
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
def test_missing_row_returns_false(self, tmp_path):
|
||||
db = SessionDB(db_path=tmp_path / "state.db")
|
||||
try:
|
||||
assert db.set_session_worktree("nope", "/wt") is False
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
class TestGitIsRepo:
|
||||
def test_true_for_repo(self, tmp_path):
|
||||
from tui_gateway import server
|
||||
|
||||
repo = str(tmp_path / "repo")
|
||||
_init_git_repo(repo)
|
||||
assert server._git_is_repo(repo) is True
|
||||
|
||||
def test_true_for_repo_without_commits(self, tmp_path):
|
||||
# A freshly `git init`-ed repo (no commits, no current branch) is still a
|
||||
# repo — this is the case a branch-name probe would get wrong.
|
||||
from tui_gateway import server
|
||||
|
||||
repo = str(tmp_path / "fresh")
|
||||
_init_git_repo(repo, with_commit=False)
|
||||
assert server._git_is_repo(repo) is True
|
||||
|
||||
def test_false_for_plain_dir(self, tmp_path):
|
||||
from tui_gateway import server
|
||||
|
||||
plain = str(tmp_path / "plain")
|
||||
os.makedirs(plain)
|
||||
assert server._git_is_repo(plain) is False
|
||||
|
||||
def test_false_for_empty_or_missing(self, tmp_path):
|
||||
from tui_gateway import server
|
||||
|
||||
assert server._git_is_repo("") is False
|
||||
assert server._git_is_repo(str(tmp_path / "does-not-exist")) is False
|
||||
|
||||
|
||||
class TestCreateSessionWorktree:
|
||||
def test_creates_worktree_and_persists_row(self, tmp_path, monkeypatch):
|
||||
from tui_gateway import server
|
||||
|
||||
repo = str(tmp_path / "repo")
|
||||
_init_git_repo(repo)
|
||||
|
||||
db = SessionDB(db_path=tmp_path / "state.db")
|
||||
monkeypatch.setattr(server, "_db", db)
|
||||
try:
|
||||
session = {"session_key": "KEY1", "cwd": repo, "explicit_cwd": True}
|
||||
info = server._create_session_worktree(session, repo)
|
||||
|
||||
assert info is not None
|
||||
# cwd repointed into the worktree, on a hermes/ branch.
|
||||
assert ".worktrees" in info["path"]
|
||||
assert info["branch"].startswith("hermes/")
|
||||
assert os.path.isdir(info["path"])
|
||||
assert session["cwd"] == info["path"]
|
||||
assert session["worktree"] == info
|
||||
|
||||
# Row persisted EAGERLY (not lazily) with the worktree mapping.
|
||||
row = db.get_session("KEY1")
|
||||
assert row is not None
|
||||
assert row["cwd"] == info["path"]
|
||||
assert row["worktree_path"] == info["path"]
|
||||
assert row["worktree_branch"] == info["branch"]
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
def test_non_repo_returns_none(self, tmp_path, monkeypatch):
|
||||
from tui_gateway import server
|
||||
|
||||
plain = str(tmp_path / "plain")
|
||||
os.makedirs(plain)
|
||||
db = SessionDB(db_path=tmp_path / "state.db")
|
||||
monkeypatch.setattr(server, "_db", db)
|
||||
try:
|
||||
session = {"session_key": "KEY2", "cwd": plain, "explicit_cwd": True}
|
||||
assert server._create_session_worktree(session, plain) is None
|
||||
# cwd unchanged, no row stamped with a worktree.
|
||||
assert session["cwd"] == plain
|
||||
assert "worktree" not in session
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
|
||||
class TestCleanupSessionWorktree:
|
||||
def test_removes_clean_worktree_and_clears_mapping(self, tmp_path, monkeypatch):
|
||||
from tui_gateway import server
|
||||
from hermes_cli import web_server
|
||||
|
||||
repo = str(tmp_path / "repo")
|
||||
_init_git_repo(repo, with_remote=True) # remote => unpushed detection active
|
||||
|
||||
db = SessionDB(db_path=tmp_path / "state.db")
|
||||
monkeypatch.setattr(server, "_db", db)
|
||||
try:
|
||||
session = {"session_key": "KEY3", "cwd": repo, "explicit_cwd": True}
|
||||
info = server._create_session_worktree(session, repo)
|
||||
assert info and os.path.isdir(info["path"])
|
||||
|
||||
# No commits made in the worktree => nothing unpushed => removable.
|
||||
preserved = web_server._cleanup_session_worktree(db, "KEY3")
|
||||
assert preserved is False
|
||||
assert not os.path.isdir(info["path"])
|
||||
# Mapping cleared so a repeat archive is a no-op.
|
||||
row = db.get_session("KEY3")
|
||||
assert row["worktree_path"] is None
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
def test_preserves_worktree_with_unpushed_commits(self, tmp_path, monkeypatch):
|
||||
from tui_gateway import server
|
||||
from hermes_cli import web_server
|
||||
|
||||
repo = str(tmp_path / "repo")
|
||||
_init_git_repo(repo, with_remote=True)
|
||||
|
||||
db = SessionDB(db_path=tmp_path / "state.db")
|
||||
monkeypatch.setattr(server, "_db", db)
|
||||
try:
|
||||
session = {"session_key": "KEY4", "cwd": repo, "explicit_cwd": True}
|
||||
info = server._create_session_worktree(session, repo)
|
||||
assert info and os.path.isdir(info["path"])
|
||||
|
||||
# Make an unpushed commit on the worktree branch — the guard must
|
||||
# then refuse to delete it.
|
||||
wt = info["path"]
|
||||
with open(os.path.join(wt, "work.txt"), "w", encoding="utf-8") as f:
|
||||
f.write("unpushed work\n")
|
||||
subprocess.run(["git", "-C", wt, "add", "."], check=True)
|
||||
subprocess.run(["git", "-C", wt, "commit", "-qm", "wip"], check=True)
|
||||
|
||||
preserved = web_server._cleanup_session_worktree(db, "KEY4")
|
||||
assert preserved is True
|
||||
assert os.path.isdir(wt) # still there — guard kept it
|
||||
# Mapping retained so a later (post-push) archive can retry.
|
||||
row = db.get_session("KEY4")
|
||||
assert row["worktree_path"] == wt
|
||||
finally:
|
||||
db.close()
|
||||
|
||||
def test_no_worktree_returns_none(self, tmp_path):
|
||||
from hermes_cli import web_server
|
||||
|
||||
db = SessionDB(db_path=tmp_path / "state.db")
|
||||
try:
|
||||
db.create_session("plain", source="tui")
|
||||
assert web_server._cleanup_session_worktree(db, "plain") is None
|
||||
finally:
|
||||
db.close()
|
||||
@@ -6,6 +6,7 @@ from pathlib import Path
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
DASHBOARD_RUN = REPO_ROOT / "docker" / "s6-rc.d" / "dashboard" / "run"
|
||||
MAIN_WRAPPER = REPO_ROOT / "docker" / "main-wrapper.sh"
|
||||
STAGE2_HOOK = REPO_ROOT / "docker" / "stage2-hook.sh"
|
||||
|
||||
|
||||
def test_main_wrapper_preserves_docker_workdir() -> None:
|
||||
@@ -77,3 +78,14 @@ def test_dashboard_run_does_not_derive_insecure_from_bind_host() -> None:
|
||||
assert truthy in text, (
|
||||
f"HERMES_DASHBOARD_INSECURE should accept truthy value {truthy!r}"
|
||||
)
|
||||
|
||||
|
||||
def test_stage2_hook_repairs_profiles_and_cron_ownership_on_every_boot() -> None:
|
||||
"""profiles/ and cron/ must both be reclaimed after root-context writes."""
|
||||
text = STAGE2_HOOK.read_text(encoding="utf-8")
|
||||
|
||||
assert 'if [ -d "$HERMES_HOME/profiles" ]; then' in text
|
||||
assert 'chown -R hermes:hermes "$HERMES_HOME/profiles" 2>/dev/null || true' in text
|
||||
|
||||
assert 'if [ -d "$HERMES_HOME/cron" ]; then' in text
|
||||
assert 'chown -R hermes:hermes "$HERMES_HOME/cron" 2>/dev/null || true' in text
|
||||
|
||||
@@ -373,6 +373,26 @@ class TestSkillView:
|
||||
assert result["name"] == "my-skill"
|
||||
assert "Step 1" in result["content"]
|
||||
|
||||
def test_view_skill_by_frontmatter_name_when_dir_differs(self, tmp_path):
|
||||
# The on-disk directory ("alias-dir") differs from the skill's
|
||||
# frontmatter name ("real-skill-name"). skills_list() exposes the
|
||||
# frontmatter name, so skill_view(name) must resolve it too.
|
||||
skill_dir = tmp_path / "alias-dir"
|
||||
skill_dir.mkdir(parents=True, exist_ok=True)
|
||||
(skill_dir / "SKILL.md").write_text(
|
||||
"---\n"
|
||||
"name: real-skill-name\n"
|
||||
"description: A skill whose directory name differs from its name.\n"
|
||||
"---\n\n"
|
||||
"# real-skill-name\n\n"
|
||||
"Step 1: Do the thing.\n"
|
||||
)
|
||||
with patch("tools.skills_tool.SKILLS_DIR", tmp_path):
|
||||
raw = skill_view("real-skill-name")
|
||||
result = json.loads(raw)
|
||||
assert result["success"] is True
|
||||
assert "Step 1" in result["content"]
|
||||
|
||||
def test_skill_view_applies_template_vars(self, tmp_path):
|
||||
with (
|
||||
patch("tools.skills_tool.SKILLS_DIR", tmp_path),
|
||||
|
||||
+57
-5
@@ -2111,19 +2111,23 @@ def delegate_task(
|
||||
# Per-task role beats top-level; normalise again so unknown
|
||||
# per-task values warn and degrade to leaf uniformly.
|
||||
effective_role = _normalize_role(t.get("role") or top_role)
|
||||
# Smart model routing: pick a tier-appropriate model per task by
|
||||
# its goal (no-op unless smart_model_routing.enabled and delegation
|
||||
# didn't pin a model). Cache-safe — children start fresh.
|
||||
task_creds = _route_task_creds(creds, str(t.get("goal") or ""), parent_agent)
|
||||
child = _build_child_agent(
|
||||
task_index=i,
|
||||
goal=t["goal"],
|
||||
context=t.get("context"),
|
||||
toolsets=t.get("toolsets") or toolsets,
|
||||
model=creds["model"],
|
||||
model=task_creds["model"],
|
||||
max_iterations=effective_max_iter,
|
||||
task_count=n_tasks,
|
||||
parent_agent=parent_agent,
|
||||
override_provider=creds["provider"],
|
||||
override_base_url=creds["base_url"],
|
||||
override_api_key=creds["api_key"],
|
||||
override_api_mode=creds["api_mode"],
|
||||
override_provider=task_creds["provider"],
|
||||
override_base_url=task_creds["base_url"],
|
||||
override_api_key=task_creds["api_key"],
|
||||
override_api_mode=task_creds["api_mode"],
|
||||
override_acp_command=t.get("acp_command")
|
||||
or acp_command
|
||||
or creds.get("command"),
|
||||
@@ -2569,6 +2573,54 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict:
|
||||
}
|
||||
|
||||
|
||||
def _route_task_creds(base_creds: dict, goal: str, parent_agent) -> dict:
|
||||
"""Apply smart_model_routing to one delegated task's credentials.
|
||||
|
||||
Cache-safe by construction: subagents start from a fresh context, so
|
||||
picking a per-task model never invalidates any cached prefix. Only acts
|
||||
when delegation didn't already pin a model (explicit ``delegation.model``
|
||||
wins), ``smart_model_routing`` is enabled with ``apply_to_delegation``,
|
||||
and the parent is on the Nous Portal (routing is Nous-only). Fail-open:
|
||||
returns ``base_creds`` unchanged on any miss or error, so the child
|
||||
inherits the parent model exactly as before.
|
||||
"""
|
||||
if base_creds.get("model"):
|
||||
return base_creds # explicit delegation model wins over routing
|
||||
try:
|
||||
from agent import model_router
|
||||
|
||||
rcfg = model_router.get_routing_config()
|
||||
if not rcfg.get("enabled") or not rcfg.get("apply_to_delegation", True):
|
||||
return base_creds
|
||||
decision = model_router.route(
|
||||
goal or "",
|
||||
current_model=getattr(parent_agent, "model", "") or "",
|
||||
current_provider=getattr(parent_agent, "provider", "") or "",
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("delegation routing: classification failed: %s", exc)
|
||||
return base_creds
|
||||
|
||||
if decision is None:
|
||||
return base_creds # no-op / stays on parent model
|
||||
|
||||
logger.info(
|
||||
"delegation routing: tier=%s → %s (%s)",
|
||||
decision.tier, decision.model, decision.provider,
|
||||
)
|
||||
routed = dict(base_creds)
|
||||
routed.update(
|
||||
{
|
||||
"model": decision.model,
|
||||
"provider": decision.provider,
|
||||
"base_url": decision.base_url,
|
||||
"api_key": decision.api_key,
|
||||
"api_mode": decision.api_mode,
|
||||
}
|
||||
)
|
||||
return routed
|
||||
|
||||
|
||||
def _load_config() -> dict:
|
||||
"""Load delegation config from CLI_CONFIG or persistent config.
|
||||
|
||||
|
||||
+12
-1
@@ -1032,10 +1032,21 @@ def skill_view(
|
||||
_record(None, categorized_path.with_suffix(".md"))
|
||||
|
||||
# Strategy 2: recursive by directory name (catches nested skills
|
||||
# like "foundations/runtime/explore-codebase" called by bare name).
|
||||
# like "foundations/runtime/explore-codebase" called by bare name),
|
||||
# plus frontmatter `name:` lookup. `skills_list()` exposes the
|
||||
# frontmatter name, so `skill_view(name)` must accept it too even
|
||||
# when the on-disk directory is a shorter category/alias.
|
||||
for found_skill_md in iter_skill_index_files(search_dir, "SKILL.md"):
|
||||
if found_skill_md.parent.name == name:
|
||||
_record(found_skill_md.parent, found_skill_md)
|
||||
continue
|
||||
try:
|
||||
fm_content = found_skill_md.read_text(encoding="utf-8")
|
||||
fm, _ = _parse_frontmatter(fm_content)
|
||||
except Exception:
|
||||
fm = {}
|
||||
if fm.get("name") == name:
|
||||
_record(found_skill_md.parent, found_skill_md)
|
||||
|
||||
# Strategy 3: legacy flat <name>.md files anywhere under the dir.
|
||||
for found_md in search_dir.rglob(f"{name}.md"):
|
||||
|
||||
@@ -1099,29 +1099,6 @@ def _git_branch_for_cwd(cwd: str) -> str:
|
||||
return ""
|
||||
|
||||
|
||||
def _git_is_repo(cwd: str) -> bool:
|
||||
"""Return True when *cwd* is inside a git working tree.
|
||||
|
||||
Used to gate the desktop sidebar's "new session in a worktree" affordance:
|
||||
the fork icon only appears for workspaces that are real git repos. Unlike a
|
||||
branch-name probe, this is unambiguous for a freshly-`git init`-ed repo with
|
||||
no commits yet (which has no current branch).
|
||||
"""
|
||||
if not cwd:
|
||||
return False
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["git", "-C", cwd, "rev-parse", "--is-inside-work-tree"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=1.5,
|
||||
check=False,
|
||||
)
|
||||
return result.returncode == 0 and result.stdout.strip() == "true"
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _session_cwd(session: dict | None) -> str:
|
||||
if session and session.get("cwd"):
|
||||
return str(session["cwd"])
|
||||
@@ -1194,81 +1171,6 @@ def _ensure_session_db_row(session: dict) -> None:
|
||||
pass
|
||||
|
||||
|
||||
def _create_session_worktree(session: dict, repo_cwd: str) -> dict | None:
|
||||
"""Create a git worktree for a desktop "new session in a worktree" request.
|
||||
|
||||
Reuses the same worktree machinery as ``hermes --worktree --tui`` (defined in
|
||||
``cli.py``). On success:
|
||||
* a fresh worktree is created under ``<repo>/.worktrees/hermes-<id>`` on a
|
||||
``hermes/hermes-<id>`` branch,
|
||||
* ``session["cwd"]`` is repointed at the worktree (so the agent + all its
|
||||
tools run inside it),
|
||||
* ``session["worktree"]`` holds the ``{path, branch, repo_root}`` metadata,
|
||||
* the session's DB row is persisted **eagerly** (unlike normal drafts) and
|
||||
stamped with the worktree mapping, so the archive handler can find and
|
||||
remove the worktree later — even after a backend restart.
|
||||
|
||||
Returns the worktree info dict, or ``None`` if *repo_cwd* is not a git repo
|
||||
or worktree creation failed (caller falls back to the plain workspace cwd).
|
||||
"""
|
||||
if not repo_cwd or not _git_is_repo(repo_cwd):
|
||||
return None
|
||||
try:
|
||||
from cli import _setup_worktree
|
||||
except Exception:
|
||||
logger.warning("worktree helpers unavailable; falling back to plain cwd", exc_info=True)
|
||||
return None
|
||||
|
||||
try:
|
||||
# Resolve the repo root for the requested workspace explicitly (rather
|
||||
# than relying on the gateway's process CWD) so the worktree is created
|
||||
# against the folder the user actually picked.
|
||||
result = subprocess.run(
|
||||
["git", "-C", repo_cwd, "rev-parse", "--show-toplevel"],
|
||||
capture_output=True, text=True, timeout=5, check=False,
|
||||
)
|
||||
repo_root = result.stdout.strip() if result.returncode == 0 else None
|
||||
if not repo_root:
|
||||
return None
|
||||
wt_info = _setup_worktree(repo_root=repo_root)
|
||||
except Exception:
|
||||
logger.warning("failed to create session worktree", exc_info=True)
|
||||
return None
|
||||
|
||||
if not wt_info or not wt_info.get("path"):
|
||||
return None
|
||||
|
||||
session["cwd"] = wt_info["path"]
|
||||
session["explicit_cwd"] = True
|
||||
session["worktree"] = wt_info
|
||||
_register_session_cwd(session)
|
||||
|
||||
# A worktree is an explicit, heavy action (unlike a blank draft), so persist
|
||||
# the row eagerly and stamp the worktree mapping. create_session is
|
||||
# INSERT-OR-IGNORE, so the later lazy create + the AIAgent's own insert
|
||||
# remain no-ops.
|
||||
key = session.get("session_key")
|
||||
db = _get_db()
|
||||
if key and db is not None:
|
||||
try:
|
||||
db.create_session(
|
||||
key,
|
||||
source="tui",
|
||||
model=_resolve_model(),
|
||||
cwd=wt_info["path"],
|
||||
)
|
||||
db.set_session_worktree(
|
||||
key,
|
||||
wt_info["path"],
|
||||
wt_info.get("branch"),
|
||||
wt_info.get("repo_root"),
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("failed to persist worktree session row", exc_info=True)
|
||||
|
||||
return wt_info
|
||||
|
||||
|
||||
def _set_session_cwd(session: dict, cwd: str) -> str:
|
||||
resolved = os.path.abspath(os.path.expanduser(str(cwd)))
|
||||
if not os.path.isdir(resolved):
|
||||
@@ -3461,19 +3363,6 @@ def _inflight_snapshot(session: dict) -> dict | None:
|
||||
# ── Methods: session ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
@method("git.is_repo")
|
||||
def _(rid, params: dict) -> dict:
|
||||
"""Report whether a directory is a git repository.
|
||||
|
||||
Desktop sidebar uses this to gate the per-workspace "new session in a
|
||||
worktree" fork icon — it only renders for workspaces that are real repos.
|
||||
Accepts an explicit ``cwd`` (a workspace path); falls back to the session /
|
||||
launch directory resolution used elsewhere.
|
||||
"""
|
||||
cwd = _completion_cwd(params)
|
||||
return _ok(rid, {"cwd": cwd, "is_repo": _git_is_repo(cwd)})
|
||||
|
||||
|
||||
@method("session.create")
|
||||
def _(rid, params: dict) -> dict:
|
||||
sid = uuid.uuid4().hex[:8]
|
||||
@@ -3536,16 +3425,6 @@ def _(rid, params: dict) -> dict:
|
||||
"transport": current_transport() or _stdio_transport,
|
||||
}
|
||||
_register_session_cwd(_sessions[sid])
|
||||
|
||||
# "New session in a worktree" (desktop sidebar fork icon): when the client
|
||||
# asks for a worktree and the chosen workspace is a git repo, create an
|
||||
# isolated worktree and run the session inside it. On failure we fall back to
|
||||
# the plain workspace cwd (the helper returns None and logs).
|
||||
worktree_info = None
|
||||
if bool(params.get("worktree")) and explicit_cwd:
|
||||
worktree_info = _create_session_worktree(
|
||||
_sessions[sid], os.path.abspath(os.path.expanduser(raw_cwd))
|
||||
)
|
||||
# NOTE: we intentionally do NOT persist a DB row here. Every TUI/desktop
|
||||
# launch (and every "New agent" / draft) opens a session here just to paint
|
||||
# the composer, so eagerly creating a row left an "Untitled" empty session
|
||||
@@ -3580,8 +3459,6 @@ def _(rid, params: dict) -> dict:
|
||||
"cwd": _sessions[sid]["cwd"],
|
||||
"branch": _git_branch_for_cwd(_sessions[sid]["cwd"]),
|
||||
"lazy": True,
|
||||
"worktree": bool(worktree_info),
|
||||
"worktree_path": worktree_info["path"] if worktree_info else None,
|
||||
"desktop_contract": DESKTOP_BACKEND_CONTRACT,
|
||||
"profile_name": _current_profile_name(),
|
||||
},
|
||||
|
||||
@@ -381,8 +381,8 @@ export default function WebhooksPage() {
|
||||
<div className="flex flex-col gap-1">
|
||||
<span className="font-medium">Webhook platform disabled</span>
|
||||
<span className="text-muted-foreground">
|
||||
The webhook platform must be enabled in your messaging settings
|
||||
before you can create subscriptions. Enable it, then return to
|
||||
The webhook platform must be enabled on the Channels page before
|
||||
you can create subscriptions. Enable it there, then return to
|
||||
this page.
|
||||
</span>
|
||||
</div>
|
||||
|
||||
@@ -0,0 +1,114 @@
|
||||
---
|
||||
title: Smart Model Routing
|
||||
description: Auto-pick a tier-appropriate model per request without breaking your prompt cache.
|
||||
sidebar_label: Smart Model Routing
|
||||
sidebar_position: 9
|
||||
---
|
||||
|
||||
# Smart Model Routing
|
||||
|
||||
Smart model routing is Hermes' take on a Cursor-style **"Auto"** model picker:
|
||||
a cheap classifier reads an incoming request, labels how much capability it
|
||||
needs (`light` / `standard` / `heavy`), and runs it on a model you've mapped to
|
||||
that tier. Hard tasks get a frontier model; trivial ones get something small
|
||||
and fast.
|
||||
|
||||
It is a **Nous Portal feature**: every tier runs on the
|
||||
[Nous Portal](https://portal.nousresearch.com), which fronts the frontier
|
||||
models across vendors (`anthropic/…`, `openai/…`, `google/…`, `x-ai/…`) behind
|
||||
a single credential — so one Portal key covers every tier. Routing only
|
||||
engages when your active model is itself on the Nous Portal; if you're on any
|
||||
other provider it stays out of the way entirely (it never moves you onto Nous).
|
||||
|
||||
It is **off by default**, and when on it is **prompt-cache-safe by design**.
|
||||
|
||||
## When does it route?
|
||||
|
||||
This is the part that makes it different from a naive "switch the model every
|
||||
turn" router. Hermes' per-conversation prompt caching is
|
||||
sacred: swapping the main model mid-conversation throws away the cached prefix
|
||||
and re-pays full input price on the new model — which, in a long thread, can
|
||||
cost *more* than it saves. So routing only ever happens where there is **no
|
||||
cached prefix to invalidate**:
|
||||
|
||||
| Where | What it does | Cache impact |
|
||||
|-------|--------------|--------------|
|
||||
| **Session start** | Classifies the first message of a *fresh* session and picks the model **before the first API call**. | None — nothing is cached yet. |
|
||||
| **Delegation** | Classifies each `delegate_task` subtask's goal and picks the subagent's model. | None — subagents start from fresh context. |
|
||||
|
||||
It does **not** swap your main model mid-conversation. That remains the job of
|
||||
the explicit [`/model`](../../reference/slash-commands.md) command (which
|
||||
deliberately resets the cache). Resumed sessions are never re-routed.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
You need the **Nous Portal** configured as a provider (`hermes auth add nous`
|
||||
or `hermes model` → Nous Portal) and your main model running on it. Routing is
|
||||
Nous-only — it stays inert on every other provider.
|
||||
|
||||
## Enabling it
|
||||
|
||||
Add a `smart_model_routing` block to `~/.hermes/config.yaml`. Tiers are just
|
||||
Nous Portal model ids:
|
||||
|
||||
```yaml
|
||||
smart_model_routing:
|
||||
enabled: true
|
||||
apply_to_sessions: true # route at the start of a fresh session
|
||||
apply_to_delegation: true # route delegated subtasks by their goal
|
||||
tiers:
|
||||
light: google/gemini-3.5-flash # fast + cheap
|
||||
standard: "" # empty = stay on your main model
|
||||
heavy: anthropic/claude-opus-4.8 # frontier
|
||||
default_tier: standard # used when the classifier can't be reached
|
||||
min_tier: "" # set to "standard" to forbid the light tier
|
||||
announce: true # print the routing decision
|
||||
|
||||
# Point the picker at a small, fast Portal model — it runs once per fresh
|
||||
# session and per delegated subtask, so an expensive classifier defeats the
|
||||
# purpose.
|
||||
auxiliary:
|
||||
routing_classifier:
|
||||
provider: nous
|
||||
model: google/gemini-3.5-flash
|
||||
```
|
||||
|
||||
### Tiers
|
||||
|
||||
There are three ordered tiers — `light`, `standard`, `heavy`. Each maps to a
|
||||
**Nous Portal model id**; the provider is always Nous, so you only specify the
|
||||
model. Credentials (`base_url`, `api_key`, `api_mode`) resolve automatically
|
||||
from your Nous credential, exactly like [`delegation.model`](./delegation.md).
|
||||
Leave a tier empty to mean **"stay on the current/parent model"** — that's the
|
||||
natural baseline for `standard`.
|
||||
|
||||
### The classifier
|
||||
|
||||
The picker runs through the `auxiliary.routing_classifier` task (see
|
||||
[Auxiliary Models](../configuration.md#auxiliary-models)). It sends the request
|
||||
to the configured Portal model and asks for a one-word tier label. It is
|
||||
**fail-open**: if the classifier is unreachable, slow, or returns garbage,
|
||||
Hermes falls back to `default_tier` and never wedges your turn.
|
||||
|
||||
## Tuning
|
||||
|
||||
- **`min_tier`** is a quality-first guardrail. Set it to `standard` to forbid
|
||||
the `light` tier entirely, so the router can upgrade but never downgrade below
|
||||
your floor. Empty means no floor.
|
||||
- **`default_tier`** is where requests land when classification fails. Keep it
|
||||
at `standard` (or higher) so a flaky classifier degrades toward quality.
|
||||
- The classifier is told to **bias toward the higher tier when unsure** — the
|
||||
most common complaint about auto-routers is picking a weak model for a hard
|
||||
task, so the default leans conservative.
|
||||
|
||||
## Relationship to other model controls
|
||||
|
||||
| Feature | What it controls |
|
||||
|---------|------------------|
|
||||
| `smart_model_routing` | Auto-picks a tier model at session start / delegation. |
|
||||
| [`/model`](../../reference/slash-commands.md) | Manual, explicit switch for the current session (always wins; resets cache). |
|
||||
| [`fallback_providers`](./fallback-providers.md) | Failover when a model **errors** (rate limit, outage) — not task-based. |
|
||||
| [`delegation.model`](./delegation.md) | Pins a fixed model for all subagents. An explicit pin **beats** routing. |
|
||||
|
||||
An explicit `delegation.model` always wins over delegation routing, and an
|
||||
explicit `/model` always wins over session routing.
|
||||
Reference in New Issue
Block a user